From 318fe89c05e8a8cc2c740474c1cbd8f6f4f26c05 Mon Sep 17 00:00:00 2001 From: Florian Date: Wed, 7 Oct 2026 01:45:07 +0200 Subject: [PATCH 1/4] feat(eval): identify adapted search captures and provenance --- scripts/discovery-eval/capture.test.ts | 48 ++++++++++++++++++++++++++ scripts/discovery-eval/capture.ts | 40 +++++++++++++++++++++ scripts/record-search-eval.mts | 18 +++++++--- 3 files changed, 102 insertions(+), 4 deletions(-) create mode 100644 scripts/discovery-eval/capture.test.ts create mode 100644 scripts/discovery-eval/capture.ts diff --git a/scripts/discovery-eval/capture.test.ts b/scripts/discovery-eval/capture.test.ts new file mode 100644 index 000000000..409d1349a --- /dev/null +++ b/scripts/discovery-eval/capture.test.ts @@ -0,0 +1,48 @@ +import { describe, expect, it } from "vitest"; +import { captureMetadata } from "./capture.js"; + +const revision = "a".repeat(40); +const now = new Date("2026-10-07T00:00:00Z"); + +describe("search capture provenance", () => { + it("identifies selected adapted API fields and leaves unobserved revisions unknown", () => { + expect(captureMetadata("https://openmapx.com/", revision, now)).toEqual({ + protocolVersion: 1, + layer: "adapted-api", + representation: "selected-fields", + recordedAt: "2026-10-07T00:00:00.000Z", + captureCodeRevision: revision, + captureWorkingTreeDirty: false, + apiOrigin: "https://openmapx.com", + deploymentRevision: null, + sourceRevisions: null, + upstreamPayloadCaptured: false, + }); + }); + + it.each([ + "https://user:secret@example.com", + "https://example.com?key=secret", + "https://example.com#secret", + "file:///tmp/secret", + "not-a-url", + ])("rejects unsafe capture base %s without including its input in the error", (api) => { + expect(() => captureMetadata(api, revision, now)).toThrow("Capture API base"); + try { + captureMetadata(api, revision, now); + } catch (error) { + expect(String(error)).not.toContain("secret"); + expect(String(error)).not.toContain(api); + } + }); + + it("supports local capture without claiming that revision was deployed", () => { + const result = captureMetadata("http://localhost:3001/", null, now); + expect(result.captureCodeRevision).toBeNull(); + expect( + captureMetadata("http://localhost:3001", revision, now, true).captureWorkingTreeDirty, + ).toBe(true); + expect(result.deploymentRevision).toBeNull(); + expect(result.apiOrigin).toBe("http://localhost:3001"); + }); +}); diff --git a/scripts/discovery-eval/capture.ts b/scripts/discovery-eval/capture.ts new file mode 100644 index 000000000..9c10dff6f --- /dev/null +++ b/scripts/discovery-eval/capture.ts @@ -0,0 +1,40 @@ +/** Provenance for filtered API recordings; never represents raw upstream data. */ +export function captureMetadata( + api: string, + revision: string | null, + now = new Date(), + workingTreeDirty = false, +) { + let base: URL; + try { + base = new URL(api); + } catch { + throw new Error( + "Capture API base must be a public HTTP(S) URL without credentials or parameters", + ); + } + if ( + !["http:", "https:"].includes(base.protocol) || + base.username || + base.password || + base.search || + base.hash + ) { + // Do not include the input: rejected URLs can contain secrets. + throw new Error( + "Capture API base must be a public HTTP(S) URL without credentials or parameters", + ); + } + return { + protocolVersion: 1 as const, + layer: "adapted-api" as const, + representation: "selected-fields" as const, + recordedAt: now.toISOString(), + captureCodeRevision: revision, + captureWorkingTreeDirty: workingTreeDirty, + apiOrigin: base.origin, + deploymentRevision: null, + sourceRevisions: null, + upstreamPayloadCaptured: false, + }; +} diff --git a/scripts/record-search-eval.mts b/scripts/record-search-eval.mts index 995a97d96..340e5b158 100644 --- a/scripts/record-search-eval.mts +++ b/scripts/record-search-eval.mts @@ -1,10 +1,11 @@ /** - * Records the provider responses behind each search-eval case, so the ranking + * Records selected adapted API responses behind each search-eval case, so the ranking * eval (packages/core/src/utils/__tests__/search-eval) runs offline and * deterministically. Needs a running API: * * pnpm search-eval:record [--api http://localhost:3001] [case-id …] * + * These are not raw upstream payloads. Each recording includes capture provenance. * Re-record after changing a provider, the geocoding chain, or a case's query * or location; the fixtures are what the eval scores. */ @@ -16,6 +17,7 @@ import { EVAL_CASES, type EvalCase, } from "../packages/core/src/utils/__tests__/search-eval/cases.js"; +import { captureMetadata } from "./discovery-eval/capture.js"; const FIXTURE_DIR = join( dirname(fileURLToPath(import.meta.url)), @@ -56,7 +58,13 @@ function pick(item: T, fields: readonly string[]): Partial ) as Partial; } -async function recordCase(evalCase: EvalCase) { +function provenance() { + const revision = execFileSync("git", ["rev-parse", "HEAD"], { encoding: "utf8" }).trim(); + const dirty = execFileSync("git", ["status", "--porcelain"], { encoding: "utf8" }).trim() !== ""; + return captureMetadata(api, revision, new Date(), dirty); +} + +async function recordCase(evalCase: EvalCase, capture: ReturnType) { const [lng, lat] = evalCase.center; const near = { lat: lat.toFixed(2), lng: lng.toFixed(2) }; const q = evalCase.query; @@ -88,6 +96,7 @@ async function recordCase(evalCase: EvalCase) { }), ]); return { + capture: { ...capture, recordedAt: new Date().toISOString() }, autocomplete: autocomplete.map((item) => pick(item, PLACE_FIELDS)), aggregate: aggregate.suggestions.map((item) => pick(item, PLACE_FIELDS)), brands: brands.matches.map((item) => @@ -98,6 +107,7 @@ async function recordCase(evalCase: EvalCase) { } async function main() { + const capture = provenance(); // Validate and snapshot the checkout before requests or writes. mkdirSync(FIXTURE_DIR, { recursive: true }); const langs = new Set(EVAL_CASES.map((evalCase) => evalCase.lang)); const chipTranslations: Record = {}; @@ -117,12 +127,12 @@ async function main() { }); writeFileSync( join(FIXTURE_DIR, "_shared.json"), - `${JSON.stringify({ chipTranslations, integrationCategories }, null, 1)}\n`, + `${JSON.stringify({ capture, chipTranslations, integrationCategories }, null, 1)}\n`, ); for (const evalCase of EVAL_CASES) { if (only.size > 0 && !only.has(evalCase.id)) continue; - const recorded = await recordCase(evalCase); + const recorded = await recordCase(evalCase, capture); writeFileSync( join(FIXTURE_DIR, `${evalCase.id}.json`), `${JSON.stringify(recorded, null, 1)}\n`, From 0ebfcc185ec5e8df25d6507e6e1cd84ab4a836b6 Mon Sep 17 00:00:00 2001 From: Florian Date: Wed, 7 Oct 2026 01:57:00 +0200 Subject: [PATCH 2/4] feat(eval): report versioned discovery and navigation evidence --- package.json | 4 +- scripts/discovery-eval/catalog.ts | 142 ++++++++++++++++ scripts/discovery-eval/report.test.ts | 124 ++++++++++++++ scripts/discovery-eval/report.ts | 224 ++++++++++++++++++++++++++ scripts/discovery-eval/run.ts | 129 +++++++++++++++ scripts/discovery-eval/tsconfig.json | 11 ++ 6 files changed, 633 insertions(+), 1 deletion(-) create mode 100644 scripts/discovery-eval/catalog.ts create mode 100644 scripts/discovery-eval/report.test.ts create mode 100644 scripts/discovery-eval/report.ts create mode 100644 scripts/discovery-eval/run.ts create mode 100644 scripts/discovery-eval/tsconfig.json diff --git a/package.json b/package.json index 4d8aab8de..bf4dec7d0 100644 --- a/package.json +++ b/package.json @@ -8,7 +8,7 @@ "lint": "biome check . && prettier --check \"**/*.{md,yml,yaml}\" && pnpm check-translations", "format": "biome format --write . && prettier --write \"**/*.{md,yml,yaml}\"", "clean": "turbo clean && rm -rf node_modules", - "check-types": "turbo run check-types", + "check-types": "turbo run check-types && pnpm discovery-eval:check-types", "check-dead-code": "knip --no-progress --reporter compact && pnpm check-unused-exports", "check-unused-exports": "knip --no-progress --reporter compact --workspace apps/ops-agent --workspace apps/transitous-runner --include exports,types --exclude files,dependencies", "check-duplicates": "jscpd --config .jscpd.json --cross-formats js-ts --no-tips apps packages integrations services scripts", @@ -43,6 +43,8 @@ "check-license-metadata": "pnpm -C packages/cli exec tsx ../../scripts/check-license-metadata.ts", "check-air-quality-release-gates": "node --disable-warning=MODULE_TYPELESS_PACKAGE_JSON --experimental-strip-types scripts/check-air-quality-release-gates.ts", "bench-air-quality": "pnpm -C packages/cli exec tsx ../../scripts/bench-air-quality.ts", + "discovery-eval:check-types": "tsc -p scripts/discovery-eval/tsconfig.json", + "discovery-eval": "pnpm -C packages/cli exec tsx ../../scripts/discovery-eval/run.ts", "search-eval:record": "pnpm -C packages/cli exec tsx ../../scripts/record-search-eval.mts", "check-toolchain-pins": "pnpm -C packages/cli exec tsx ../../scripts/check-toolchain-pins.ts", "check:policy": "pnpm check-legal-tables && pnpm check-legal-updated && pnpm check-data-flows && pnpm check-subject-data && pnpm check-license-metadata && pnpm check-air-quality-release-gates && pnpm check-toolchain-pins && pnpm check-feed-ids && pnpm check-credential-keys && pnpm check-dockerfile-sync && pnpm check-image-size-dos && pnpm check-docker-context-secrets && pnpm check-ops-authority", diff --git a/scripts/discovery-eval/catalog.ts b/scripts/discovery-eval/catalog.ts new file mode 100644 index 000000000..8f92cc888 --- /dev/null +++ b/scripts/discovery-eval/catalog.ts @@ -0,0 +1,142 @@ +import { EVAL_CASES } from "../../packages/core/src/utils/__tests__/search-eval/cases.js"; +import { OVERTURE_QUALITY_BASELINE_RELEASE } from "../../services/data-manager/src/jobs/overture/eval/quality-baseline.js"; +import type { EvalCase } from "./report.js"; + +const search = "packages/core/src/utils/__tests__/search-eval"; +const navigation = "packages/core/src/navigation"; +const overture = "services/data-manager/__tests__/overture"; +const cards = "apps/web/src/components/panels/category/CategoryResultsContent.test.tsx"; + +/** Existing assertions stay authoritative; this is a catalog, not a second ranker. */ +export const CATALOG: EvalCase[] = [ + ...EVAL_CASES.map((entry) => ({ + id: `search/${entry.id}`, + layer: "client-ranking/recorded-adapted-api", + suite: `${search}/search-eval.test.ts`, + assertions: [`${entry.id}:`], + fixtures: [`${search}/fixtures/${entry.id}.json`, `${search}/fixtures/_shared.json`], + expected: entry, + ...(entry.knownGap ? { knownGap: entry.knownGap } : {}), + })), + { + id: "search/aggregate-ranking-budget", + layer: "client-ranking/recorded-adapted-api", + suite: `${search}/search-eval.test.ts`, + assertions: ["keeps the first expectation at rank 1 for most cases"], + expected: { hitAt1: 0.8, reciprocalRank: 0.85 }, + }, + { + id: "search/station-synonyms-cache-order", + layer: "adapted-api/mocked-upstream", + suite: "integrations/geocoding/__tests__/forward-ranking.test.ts", + note: "Cold/warm query order, station aliases, language/proximity isolation and provider-order fallback; mocked upstream, not provider coverage.", + }, + { + id: "search/location-cache-isolation", + layer: "adapted-api/mocked-upstream", + suite: "integrations/geocoding/__tests__/routes.test.ts", + }, + { + id: "identity/co-located-tenants", + layer: "conflation/synthetic", + suite: "packages/core/src/utils/__tests__/poiConflation.test.ts", + note: "Includes plural-per-address restaurants, contradictory address/phone and shared switchboard guards; does not establish real mall floor identity.", + }, + { + id: "place/partial-enrichment-loading", + layer: "ui/contract-fixtures", + suite: cards, + note: "Partial photos/ratings, independent credits, bounded retries, stale searches and missing data; no real-provider availability or production latency claim.", + }, + { + id: "coverage/overture-reviewed-gate-contract", + layer: "dataset-gate/unit-fixtures", + suite: `${overture}/eval/quality-gate.test.ts`, + expected: { reviewedRelease: OVERTURE_QUALITY_BASELINE_RELEASE, resultWindow: 50 }, + note: "Berlin/Aachen/Monschau/Maastricht anchor gate contracts; this command does not query a deployed dataset or measure current regional recall.", + }, + { + id: "coverage/overture-metric-contract", + layer: "dataset-gate/unit-fixtures", + suite: `${overture}/eval/metrics.test.ts`, + }, + { + id: "coverage/overture-search-quality-contract", + layer: "dataset-gate/unit-fixtures", + suite: `${overture}/eval/search-quality.test.ts`, + }, + { + id: "navigation/transit-recovery-gps-gaps", + layer: "navigation-engine/synthetic-replay", + suite: `${navigation}/mobileReplay.test.ts`, + fixtures: ["ground-basic", "transit-basic", "transit-tunnel", "transit-transfer"].map( + (name) => `${navigation}/__fixtures__/mobile/${name}.json`, + ), + note: "Deterministic transfers, serialization/recovery and GPS gap confidence. Not installed shell/device verification.", + }, + { + id: "navigation/ground-off-route-arrival", + layer: "navigation-engine/synthetic", + suite: `${navigation}/__tests__/processFix.test.ts`, + }, + { + id: "navigation/alternative-route", + layer: "navigation-engine/synthetic", + suite: `${navigation}/fasterRoute.test.ts`, + }, + { + id: "manual/urban-rural-browsing", + layer: "ui/live-manual", + unavailable: + "Repeat the map comparison baseline at fixed viewport, coordinates, zoom and provider/style/source revisions; attach external capture manifest and judgments.", + }, + { + id: "manual/closed-missing-businesses", + layer: "dataset/live-manual", + unavailable: + "Independently verify a stratified regional sample of missing/closed businesses against exact OSM/Overture generations; no current closure ground truth is captured here.", + }, + { + id: "manual/mall-tenant-floor-identity", + layer: "dataset/live-manual", + unavailable: + "Independently judge tenants, branches, entrances and floors; synthetic conflation guards do not establish regional coverage.", + }, + { + id: "manual/provider-performance", + layer: "adapted-api/live-manual", + unavailable: + "Declare request/latency budgets and repeat cold/warm captures in both query orders, with explicit region/configuration/source revisions.", + }, + { + id: "unavailable/installed-navigation", + layer: "installed-device", + unavailable: + "Installed navigation composition and device evidence pending #398; shared-engine replays are separate.", + }, + { + id: "unavailable/offline-place-search", + layer: "installed-device/offline", + unavailable: + "Offline place/address search pending #403; downloaded map tiles are not offline search.", + }, + { + id: "unavailable/offline-rerouting", + layer: "installed-device/offline", + unavailable: + "Android offline routing/rerouting prototype pending #404; no airplane-mode engine measurement in this corpus.", + }, +]; + +export const INPUT_FILES = [ + ...new Set([ + ...CATALOG.flatMap((entry) => [ + ...(entry.fixtures ?? []), + ...(entry.suite ? [entry.suite] : []), + ]), + `${search}/cases.ts`, + "services/data-manager/src/jobs/overture/eval/quality-baseline.ts", + "apps/web/public/styles/openmapx-streets.json", + "apps/web/public/styles/openmapx-dark.json", + ]), +].sort(); diff --git a/scripts/discovery-eval/report.test.ts b/scripts/discovery-eval/report.test.ts new file mode 100644 index 000000000..fbb239aa9 --- /dev/null +++ b/scripts/discovery-eval/report.test.ts @@ -0,0 +1,124 @@ +import { describe, expect, it } from "vitest"; +import { compareReports, createReport, type EvalCase, type EvidenceMetadata } from "./report.js"; + +const cases: EvalCase[] = [ + { id: "station", layer: "client-ranking", suite: "search.test.ts", assertions: ["station rank"] }, + { id: "offline-search", layer: "installed-device", unavailable: "Not implemented" }, +]; +const metadata: EvidenceMetadata = { + appRevision: "a".repeat(40), + workingTreeDirty: false, + inputHashes: { "fixture/station.json": "1" }, + captureProvenance: { legacyFixtures: 1, adaptedApiCaptures: 0 }, +}; +function input(status = "passed", name = "station rank") { + return { + success: status === "passed", + testResults: [ + { + name: "/repo/search.test.ts", + status: status === "failed" ? "failed" : "passed", + assertionResults: [ + { fullName: name, status, failureMessages: ["potentially sensitive stack"] }, + ], + }, + ], + }; +} + +describe("versioned discovery evidence reports", () => { + it("keeps unsupported capabilities unavailable and omits failure details", () => { + const report = createReport(input(), cases, metadata, "/repo"); + expect(report.cases[0]).toMatchObject({ id: "station", status: "passed", passed: 1 }); + expect(report.cases[1]).toMatchObject({ id: "offline-search", status: "unavailable" }); + expect(JSON.stringify(report)).not.toContain("sensitive"); + }); + + it.each(["skipped", "pending", "todo"])("does not turn %s assertions into success", (status) => { + expect(createReport(input(status), cases, metadata, "/repo").cases[0].status).toBe( + "unavailable", + ); + }); + + it("does not pass missing assertions or collection failures", () => { + expect( + createReport(input("passed", "other test"), cases, metadata, "/repo").cases[0].status, + ).toBe("unavailable"); + const failedCollection = input(); + failedCollection.testResults[0].status = "failed"; + failedCollection.testResults[0].assertionResults = []; + expect(createReport(failedCollection, cases, metadata, "/repo").cases[0].status).toBe("failed"); + }); + + it("keeps an independently passing case intact when a different assertion fails", () => { + const evidence = input(); + evidence.success = false; + evidence.testResults[0].status = "failed"; + evidence.testResults[0].assertionResults.push({ + fullName: "different case", + status: "failed", + failureMessages: [], + }); + const report = createReport(evidence, cases, metadata, "/repo"); + expect(report.cases[0].status).toBe("passed"); + expect(report.runSucceeded).toBe(false); + }); + + it("reports guarded known gaps separately from passing semantic expectations", () => { + const report = createReport( + input(), + [{ ...cases[0], knownGap: "Expected station missing in recorded data" }], + metadata, + "/repo", + ); + expect(report.cases[0]).toMatchObject({ + status: "known-gap", + reason: "Expected station missing in recorded data", + }); + expect(report.runSucceeded).toBe(true); + }); + + it("detects a failed expected result independently of changing input fingerprints", () => { + const before = createReport(input(), cases, metadata, "/repo"); + const after = createReport( + input("failed"), + cases, + { ...metadata, inputHashes: { "fixture/station.json": "2" } }, + "/repo", + ); + expect(compareReports(before, after)).toMatchObject({ + regressions: ["station"], + improvements: [], + changedInputs: ["fixture/station.json"], + appOnlyComparison: false, + }); + }); + + it("rejects malformed baseline reports and mismatched case sets", () => { + const report = createReport(input(), cases, metadata, "/repo"); + expect(() => compareReports({} as typeof report, report)).toThrow("report"); + expect(() => compareReports({ ...report, cases: [] }, report)).toThrow("case set"); + expect(() => + compareReports({ ...report, cases: [report.cases[0], report.cases[0]] }, report), + ).toThrow("report"); + }); + + it("rejects comparisons across protocol, catalog or budget changes", () => { + const before = createReport(input(), cases, metadata, "/repo"); + expect(() => compareReports(before, { ...before, protocolVersion: 2 })).toThrow("protocol"); + const changed = createReport( + input(), + [{ ...cases[0], assertions: ["different budget"] }], + metadata, + "/repo", + ); + expect(() => compareReports(before, changed)).toThrow("catalog"); + }); + + it.each([null, {}, { success: true, testResults: [{ name: "/repo/search.test.ts" }] }])( + "rejects malformed test evidence %j", + (evidence) => { + expect(() => createReport(evidence, cases, metadata, "/repo")).toThrow("test evidence"); + }, + ); +}); diff --git a/scripts/discovery-eval/report.ts b/scripts/discovery-eval/report.ts new file mode 100644 index 000000000..52bf226c8 --- /dev/null +++ b/scripts/discovery-eval/report.ts @@ -0,0 +1,224 @@ +import { createHash } from "node:crypto"; +import { relative } from "node:path"; + +export interface EvalCase { + id: string; + layer: string; + suite?: string; + assertions?: string[]; + unavailable?: string; + fixtures?: string[]; + expected?: unknown; + note?: string; + knownGap?: string; +} +export interface EvidenceMetadata { + appRevision: string; + workingTreeDirty: boolean; + inputHashes: Record; + captureProvenance: { legacyFixtures: number; adaptedApiCaptures: number }; +} +export interface EvalReport { + protocolVersion: number; + catalogHash: string; + runSucceeded: boolean; + metadata: EvidenceMetadata; + cases: Array<{ + id: string; + layer: string; + status: "passed" | "failed" | "unavailable" | "known-gap"; + passed: number; + failed: number; + unavailable: number; + reason?: string; + }>; +} +interface Assertion { + fullName: string; + status: string; +} +interface TestFile { + name: string; + status: string; + assertionResults: Assertion[]; +} +const STATUSES = new Set(["passed", "failed", "skipped", "pending", "todo"]); +function record(value: unknown): value is Record { + return value !== null && typeof value === "object" && !Array.isArray(value); +} +function testEvidence(input: unknown): { success: boolean; testResults: TestFile[] } { + if (!record(input) || typeof input.success !== "boolean" || !Array.isArray(input.testResults)) { + throw new Error("Invalid test evidence"); + } + const files = input.testResults.map((file) => { + if ( + !record(file) || + typeof file.name !== "string" || + !["passed", "failed"].includes(String(file.status)) || + !Array.isArray(file.assertionResults) + ) { + throw new Error("Invalid test evidence"); + } + const assertions = file.assertionResults.map((assertion) => { + if ( + !record(assertion) || + typeof assertion.fullName !== "string" || + typeof assertion.status !== "string" || + !STATUSES.has(assertion.status) + ) { + throw new Error("Invalid test evidence"); + } + // Intentionally exclude errors, stacks, URLs, meta and test payloads. + return { fullName: assertion.fullName, status: assertion.status }; + }); + return { name: file.name, status: String(file.status), assertionResults: assertions }; + }); + return { success: input.success, testResults: files }; +} + +export function createReport( + input: unknown, + catalog: EvalCase[], + metadata: EvidenceMetadata, + root: string, +): EvalReport { + const evidence = testEvidence(input); + if (new Set(catalog.map((entry) => entry.id)).size !== catalog.length) { + throw new Error("Duplicate evaluation catalog ID"); + } + const cases = catalog.map((entry): EvalReport["cases"][number] => { + const base = { id: entry.id, layer: entry.layer, passed: 0, failed: 0, unavailable: 0 }; + if (!entry.suite) + return { + ...base, + status: "unavailable", + reason: entry.unavailable ?? "Manual evidence required", + }; + const files = evidence.testResults.filter( + (file) => relative(root, file.name).replaceAll("\\", "/") === entry.suite, + ); + const all = files.flatMap((file) => file.assertionResults); + const selected = entry.assertions?.length + ? all.filter((assertion) => + entry.assertions?.some((selector) => assertion.fullName.includes(selector)), + ) + : all; + const missing = + !files.length || + !selected.length || + entry.assertions?.some( + (selector) => !all.some((assertion) => assertion.fullName.includes(selector)), + ); + const passed = selected.filter((assertion) => assertion.status === "passed").length; + const failed = selected.filter((assertion) => assertion.status === "failed").length; + const unavailable = selected.length - passed - failed; + const status = + failed || files.some((file) => file.status === "failed" && file.assertionResults.length === 0) + ? "failed" + : missing || unavailable + ? "unavailable" + : entry.knownGap + ? "known-gap" + : "passed"; + return { + ...base, + status, + passed, + failed, + unavailable, + ...(entry.knownGap && status === "known-gap" ? { reason: entry.knownGap } : {}), + ...(missing + ? { reason: "Required suite or assertion missing" } + : unavailable + ? { reason: "Required assertions skipped, pending or todo" } + : {}), + }; + }); + return { + protocolVersion: 1, + catalogHash: createHash("sha256").update(JSON.stringify(catalog)).digest("hex"), + runSucceeded: + evidence.success && + cases.every( + (entry, index) => !catalog[index].suite || ["passed", "known-gap"].includes(entry.status), + ), + metadata, + cases, + }; +} + +function validReport(value: unknown): value is EvalReport { + if ( + !record(value) || + value.protocolVersion !== 1 || + typeof value.catalogHash !== "string" || + typeof value.runSucceeded !== "boolean" || + !record(value.metadata) || + typeof value.metadata.appRevision !== "string" || + typeof value.metadata.workingTreeDirty !== "boolean" || + !record(value.metadata.inputHashes) || + !Object.values(value.metadata.inputHashes).every((hash) => typeof hash === "string") || + !record(value.metadata.captureProvenance) || + ![ + value.metadata.captureProvenance.legacyFixtures, + value.metadata.captureProvenance.adaptedApiCaptures, + ].every((n) => Number.isInteger(n) && (n as number) >= 0) || + !Array.isArray(value.cases) + ) + return false; + return ( + value.cases.every( + (entry) => + record(entry) && + typeof entry.id === "string" && + typeof entry.layer === "string" && + ["passed", "failed", "unavailable", "known-gap"].includes(String(entry.status)) && + [entry.passed, entry.failed, entry.unavailable].every( + (n) => Number.isInteger(n) && (n as number) >= 0, + ) && + (entry.reason === undefined || typeof entry.reason === "string"), + ) && new Set(value.cases.map((entry) => entry.id)).size === value.cases.length + ); +} + +export function compareReports(before: unknown, after: unknown) { + if ( + !record(before) || + !record(after) || + typeof before.protocolVersion !== "number" || + typeof after.protocolVersion !== "number" + ) + throw new Error("Invalid evaluation report"); + if (before.protocolVersion !== after.protocolVersion) + throw new Error("Incompatible evaluation protocol versions"); + if (!validReport(before) || !validReport(after)) throw new Error("Invalid evaluation report"); + if (before.catalogHash !== after.catalogHash) + throw new Error("Incompatible evaluation catalog or budgets"); + if ( + before.cases.length !== after.cases.length || + before.cases.some((entry) => !after.cases.some((other) => entry.id === other.id)) + ) + throw new Error("Incompatible evaluation case set"); + const previous = new Map(before.cases.map((entry) => [entry.id, entry])); + const keys = new Set([ + ...Object.keys(before.metadata.inputHashes), + ...Object.keys(after.metadata.inputHashes), + ]); + const changedInputs = [...keys] + .filter((key) => before.metadata.inputHashes[key] !== after.metadata.inputHashes[key]) + .sort(); + return { + regressions: after.cases + .filter((entry) => previous.get(entry.id)?.status === "passed" && entry.status !== "passed") + .map((entry) => entry.id), + improvements: after.cases + .filter((entry) => previous.get(entry.id)?.status !== "passed" && entry.status === "passed") + .map((entry) => entry.id), + changedInputs, + appOnlyComparison: + changedInputs.length === 0 && + !before.metadata.workingTreeDirty && + !after.metadata.workingTreeDirty, + runRegression: before.runSucceeded && !after.runSucceeded, + }; +} diff --git a/scripts/discovery-eval/run.ts b/scripts/discovery-eval/run.ts new file mode 100644 index 000000000..4b61e7fbe --- /dev/null +++ b/scripts/discovery-eval/run.ts @@ -0,0 +1,129 @@ +import { execFileSync, spawnSync } from "node:child_process"; +import { createHash } from "node:crypto"; +import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { dirname, join, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; +import { CATALOG, INPUT_FILES } from "./catalog.js"; +import { compareReports, createReport, type EvalReport } from "./report.js"; + +const root = resolve(dirname(fileURLToPath(import.meta.url)), "../.."); +function options(args: string[]) { + const result = { out: join(root, ".superpowers/eval-reports/latest"), baseline: "" }; + for (let i = 0; i < args.length; i += 2) { + if (!["--out", "--baseline"].includes(args[i]) || !args[i + 1] || args[i + 1].startsWith("--")) + throw new Error("Usage: pnpm discovery-eval [--out DIRECTORY] [--baseline REPORT.json]"); + result[args[i] === "--out" ? "out" : "baseline"] = resolve(args[i + 1]); + } + return result; +} +function markdown(report: EvalReport, comparison: ReturnType | null) { + const counts = report.cases.reduce>((all, entry) => { + all[entry.status] = (all[entry.status] ?? 0) + 1; + return all; + }, {}); + return [ + "# Discovery evaluation report", + "", + `Protocol: ${report.protocolVersion}; catalog: ${report.catalogHash}`, + `Application: ${report.metadata.appRevision}; working tree dirty: ${report.metadata.workingTreeDirty}`, + `Run succeeded: ${report.runSucceeded}; case counts: ${JSON.stringify(counts)}`, + "", + "This is offline recorded/synthetic/contract evidence. Manual and installed-device results are unavailable. Assertion counts can overlap between cases; they are not independent observations or production performance metrics.", + `Search capture provenance: ${report.metadata.captureProvenance.legacyFixtures} legacy fixtures with unknown provenance; ${report.metadata.captureProvenance.adaptedApiCaptures} explicitly adapted API captures. Exact deployed source generations are unknown here.`, + "", + ...(comparison + ? [ + "## Comparison", + "", + `\`\`\`json\n${JSON.stringify(comparison, null, 2)}\n\`\`\``, + "", + "Changed inputs and dirty trees prevent an application-only comparison; comparisons do not establish causality.", + "", + ] + : []), + "| Case | Evidence layer | Status | Passed / failed / unavailable assertions | Reason |", + "| --- | --- | --- | --- | --- |", + ...report.cases.map( + (entry) => + `| ${entry.id} | ${entry.layer} | ${entry.status} | ${entry.passed} / ${entry.failed} / ${entry.unavailable} | ${entry.reason ?? ""} |`, + ), + "", + ].join("\n"); +} +function main() { + const config = options(process.argv.slice(2)); + // Snapshot before output writes; reports cannot mark their own capture as dirty. + const appRevision = execFileSync("git", ["rev-parse", "HEAD"], { + cwd: root, + encoding: "utf8", + }).trim(); + const workingTreeDirty = + execFileSync("git", ["status", "--porcelain"], { cwd: root, encoding: "utf8" }).trim() !== ""; + const inputHashes = Object.fromEntries( + INPUT_FILES.map((path) => [ + path, + createHash("sha256") + .update(readFileSync(join(root, path))) + .digest("hex"), + ]), + ); + const captureProvenance = { legacyFixtures: 0, adaptedApiCaptures: 0 }; + for (const path of INPUT_FILES.filter( + (path) => path.includes("search-eval/fixtures/") && !path.endsWith("_shared.json"), + )) { + const fixture = JSON.parse(readFileSync(join(root, path), "utf8")); + if (fixture.capture?.protocolVersion === 1 && fixture.capture?.layer === "adapted-api") + captureProvenance.adaptedApiCaptures++; + else captureProvenance.legacyFixtures++; + } + const temp = mkdtempSync(join(tmpdir(), "openmapx-discovery-")); + try { + const output = join(temp, "vitest.json"); + const suites = [...new Set(CATALOG.flatMap((entry) => (entry.suite ? [entry.suite] : [])))]; + const test = spawnSync( + "pnpm", + ["exec", "vitest", "run", ...suites, "--reporter=json", `--outputFile=${output}`], + { + cwd: root, + stdio: "inherit", + env: { ...process.env, VITEST_MAX_WORKERS: process.env.VITEST_MAX_WORKERS ?? "4" }, + }, + ); + if (test.error) throw new Error("Could not start evaluation tests"); + const report = createReport( + JSON.parse(readFileSync(output, "utf8")), + CATALOG, + { appRevision, workingTreeDirty, inputHashes, captureProvenance }, + root, + ); + const comparison = config.baseline + ? compareReports(JSON.parse(readFileSync(config.baseline, "utf8")), report) + : null; + mkdirSync(config.out, { recursive: true }); + writeFileSync( + join(config.out, "report.json"), + `${JSON.stringify({ ...report, ...(comparison ? { comparison } : {}) }, null, 2)}\n`, + ); + writeFileSync(join(config.out, "report.md"), markdown(report, comparison)); + console.log(`Evaluation report: ${join(config.out, "report.md")}`); + if ( + test.status !== 0 || + !report.runSucceeded || + comparison?.regressions.length || + comparison?.runRegression + ) + process.exitCode = 1; + } finally { + rmSync(temp, { recursive: true, force: true }); + } +} +try { + main(); +} catch { + // Inputs/exception payloads may contain secrets; no raw stacks or reporter details. + console.error( + "Evaluation failed: check arguments, readable input files and compatible baseline. No successful report is implied.", + ); + process.exitCode = 1; +} diff --git a/scripts/discovery-eval/tsconfig.json b/scripts/discovery-eval/tsconfig.json new file mode 100644 index 000000000..a3bb5722b --- /dev/null +++ b/scripts/discovery-eval/tsconfig.json @@ -0,0 +1,11 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { + "target": "ES2022", + "module": "ESNext", + "moduleResolution": "Bundler", + "noEmit": true, + "typeRoots": ["../../packages/cli/node_modules/@types"] + }, + "include": ["*.ts", "../record-search-eval.mts"] +} From 0725f76cc018764f339a27622372dffa26ee18b9 Mon Sep 17 00:00:00 2001 From: Florian Date: Wed, 7 Oct 2026 02:03:17 +0200 Subject: [PATCH 3/4] fix(eval): require complete consistent comparison evidence --- docs/docs/developer/development-setup.md | 2 + docs/docs/developer/discovery-evaluation.md | 191 ++++++++++++++++++ .../docs/developer/map-comparison-baseline.md | 5 + scripts/discovery-eval/catalog.test.ts | 27 +++ scripts/discovery-eval/catalog.ts | 1 + scripts/discovery-eval/report.test.ts | 23 +++ scripts/discovery-eval/report.ts | 32 ++- 7 files changed, 279 insertions(+), 2 deletions(-) create mode 100644 docs/docs/developer/discovery-evaluation.md create mode 100644 scripts/discovery-eval/catalog.test.ts diff --git a/docs/docs/developer/development-setup.md b/docs/docs/developer/development-setup.md index 2f8932c70..4b2a2b06e 100644 --- a/docs/docs/developer/development-setup.md +++ b/docs/docs/developer/development-setup.md @@ -221,6 +221,8 @@ Tooling notes: `web` (the Next.js app and React-bound packages, in jsdom). All test commands run from the repo root: - `pnpm test` — run the whole suite once + - `pnpm discovery-eval --out /tmp/openmapx-eval` — run the versioned offline + [discovery evaluation](discovery-evaluation.md) and save a comparison report - `pnpm test:watch` — watch mode - `pnpm test:coverage` — run with a V8 coverage report (written to `coverage/`) - `pnpm test --project web` / `--project node` — scope to one environment diff --git a/docs/docs/developer/discovery-evaluation.md b/docs/docs/developer/discovery-evaluation.md new file mode 100644 index 000000000..53c711bb9 --- /dev/null +++ b/docs/docs/developer/discovery-evaluation.md @@ -0,0 +1,191 @@ +--- +title: Discovery evaluation +description: Versioned offline discovery and navigation evidence, with a protocol for independently judged live comparisons. +--- + +# Discovery evaluation + +Use this protocol before changing discovery, place enrichment or navigation. It +connects existing semantic search fixtures, conflation guards, Overture gates, +place-card contracts and navigation replays. It does not turn synthetic tests +into evidence of current regional coverage or installed-device readiness. + +## Run and compare + +From the repository root: + +```bash +pnpm discovery-eval --out /tmp/openmapx-eval-before +# Change application code, retaining the same inputs, cases and budgets. +pnpm discovery-eval --out /tmp/openmapx-eval-after \ + --baseline /tmp/openmapx-eval-before/report.json +``` + +The default output directory is ignored `.superpowers/eval-reports/latest`. +Keep a baseline outside that directory so a later run cannot overwrite it. Each +run writes `report.json` and `report.md`, with the application commit, dirty-tree +flag, SHA-256 fingerprints of input fixtures, assertion suites, expectations and +both map styles. Temporary Vitest evidence is removed after reporting; its error +stacks and arbitrary test payloads are excluded from the saved report. + +Protocol version 1 rejects a baseline with another version, catalog, budgets or +case set or assertion inventory. Assertion names are retained as hashes, so a +shortened passing suite cannot silently replace complete baseline evidence. +Changed input fingerprints are listed separately from behavioral +regressions. `appOnlyComparison` requires unchanged fingerprints and two clean +working trees; it establishes comparability, not causality. Commit intentional +changes before capturing review evidence. A local experiment in a dirty tree is +still useful, but its HEAD does not identify all code that ran. + +The command invokes the existing suites without live provider requests. Those +suites and the report/capture regressions run through ordinary `pnpm test` in CI; +no paid API or deployed dataset is required. Exit status is nonzero for failed +runs, missing/skipped required automated assertions or a comparison regression. +Manual/unimplemented cases remain unavailable and do not block this offline gate. +A failure elsewhere in a selected suite still fails the run, even if individual +catalog cases passed. + +## Cases and evidence layers + +The versioned catalog is `scripts/discovery-eval/catalog.ts`. Its automated +expectations reuse existing tests, rather than scoring the current provider's +ordering as truth. + +| Case family | Current evidence | Required live complement | +| ------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------- | +| Stations, aliases, multilingual names, category queries and chains | Recorded adapted API inputs; independently specified expected labels, types, IDs and coordinate radii in `search-eval/cases.ts` | Verify the desired entity and each branch against an independent source before re-recording | +| Station synonyms and caches | Mocked upstream cold/warm and query-order regressions; language/proximity partitions | Repeat both query orders on an isolated local cache with a fixed provider configuration | +| Co-located tenants and branches | Synthetic conflation guards for plural-per-address categories, contradictory contacts/addresses and shared switchboards | Judge actual tenants, entrances and floors; equal address does not establish identity | +| Partial place cards | Contract fixtures for photo/rating failures, independent credits, loading, bounded retries and stale searches | Inspect missing facts and delayed requests in the browser with a pinned API revision | +| Urban/rural categories | Unit contracts for the reviewed Overture anchor gate in Aachen, Berlin, Monschau and Maastricht | Run the staged-release gate against an exact imported generation and repeat the visual baseline | +| Closed/missing businesses | Manual case, unavailable in this command | Independently dated closure/existence judgments; filtering code alone is insufficient | +| Transit transitions, recovery and GPS gaps | Synthetic shared-engine replays, including transfers and serialized recovery | Installed shell, permissions, lifecycle and actual-device evidence under #398 | +| Ground off-route behavior, arrival and alternatives | Synthetic shared-engine assertions | Repeat a real or independently recorded route with a pinned routing dataset | +| Offline place search and rerouting | Unavailable pending #403 and #404 | Network-denial or airplane-mode evidence once those capabilities exist | + +An assertion passes, fails or is unavailable. An explicitly guarded known gap is +reported as `known-gap`, rather than claiming its semantic expectation passed. +Manual cases describe the evidence still needed. Assertion counts may overlap +between cases and are not unique observations, recall values or production +performance measurements. + +## Expectations and budgets before a comparison + +Freeze the case IDs, expected entities/coordinates, assessor judgments and +acceptance rules before looking at changed results. Existing search expectations +use entity/type/label and coordinate radii; the aggregate floor is Hit@1 ≥ 0.80 +and mean reciprocal rank ≥ 0.85 for the first expectation of asserted cases. This +is an offline ranking floor, not population-wide search accuracy. + +The Overture baseline is tied to reviewed release `2026-07-22.0` and uses a +50-result window with per-case result-count, relevant-anchor recall, +known-irrelevant and known-duplicate ceilings. The offline command checks those +contracts; it does not compute today's regional metrics. Follow the +[Overture evaluation procedure](https://github.com/OpenMapX/openmapx/blob/main/services/data-manager/src/jobs/overture/eval/README.md) +for staged imports and independently labeled candidate pairs. + +For live cases, write the following budgets into the capture manifest **before** +the comparison. Retain counts and denominators, including unknown judgments. + +| Dimension | Record and judge | +| ---------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Visible discovery | At each fixed viewport/zoom, count useful labels, overlap/occlusion and independently expected missing businesses; set the acceptable crowding ceiling | +| Relevance and identity | Rank of independently expected entities; relevant/irrelevant/unjudged counts; duplicate, wrong branch, tenant and floor judgments | +| Requests and latency | Cold/warm request counts and timings with cache/provider state; declare latency and request ceilings; do not infer them from Vitest duration | +| Partial loading | Stable row/selection/focus, distinct unknown versus closed facts, independently reachable credits/actions, stale-response isolation and retry budget | +| Navigation | Expected ordered events, mode/leg transitions, recovery state, off-route decisions and alternative selection; define acceptable location/time tolerances | + +Avoid inventing a universal crowding or latency threshold from one capture. The +existing card enrichment batches eight items and limits dispatch attempts to two; +retain those contracts unless a reviewed change deliberately updates them. + +## Pin the inputs and separate the outcomes + +Use the [German map baseline](map-comparison-baseline.md) for fixed urban/rural +coordinates, zooms, searches and sheets. Keep its cases while adding new ones. +A live archive should contain a compact manifest with: + +- Protocol/catalog revision, capture timestamp and assessor identity/date. +- Application/deployment commits (separately), dirty state and style checksums. +- Region/bounding box, viewport, DPR, language, zoom, pitch, bearing and theme. +- Selected provider/capabilities, hosted versus self-hosted mode and allowlisted + non-secret settings that can affect ranking or rendering. +- Exact OSM extract date/hash, Overture release/generation and other source + generations; explicitly `null` with a reason when unavailable. +- Independently expected entities/coordinates, judgments and predeclared budgets. +- Cache state, query order, request counts/timings and artifact checksums. +- Source licenses, permitted storage/access, redaction performed and limitations. + +A fixture checksum identifies bytes, not their source date. Local Git HEAD does +not identify a remote deployment. The existing visual baseline records unknown +production revisions where they could not be observed; preserve that uncertainty. +Do not export `.env`, request headers, tokens or entire admin configuration. + +Keep three evidence layers separate when diagnosing search: + +1. **Raw upstream:** original provider payload and documented request parameters, + captured by the operator only when provider terms allow. Redact credentials + before storage; do not call a filtered response raw. +2. **Adapted API:** normalized/filtered API results with the deployment and adapter + revision, capability/configuration and response ordering. +3. **Final UI:** client-ranked rows, selection/Enter outcome, map/sheet state and + application/style revision. + +Future `pnpm search-eval:record --api http://localhost:3001 [case-id …]` +recordings identify themselves as **adapted API, selected fields**, with a capture +time, local recorder revision/dirty flag and public API origin. They explicitly +leave remote deployment and source revisions unknown and do not include upstream +payloads. Credential/query/fragment-bearing base URLs are rejected without +printing their contents. Add independently observed remote/source revisions to +an operator manifest; do not infer them from the recorder's local commit. Existing +legacy fixtures are retained with unknown capture provenance. + +## Cache and query-order isolation + +Use an isolated local Redis/database or the existing mocked tests. Never flush a +shared production cache for evaluation. Record the cache implementation/version +and state; a warm response is a different condition from a cold response. + +For station cases, run both orders on separate clean cache namespaces: +`Hauptbahnhof Neuss → Neuss Hauptbahnhof` and +`Neuss Hauptbahnhof → Hauptbahnhof Neuss`. Repeat warm, include `Hbf` aliases, +and vary language and proximity deliberately to test partitioning. Compare the +entity identity/coordinates and ordering at each evidence layer. The regression +in #389 is repaired in #392; this protocol retains the cases without replacing +that fix. Search forwarding and UI ranking are different stages. + +## Store evidence outside Git + +Keep capture archives in operator-controlled object storage or a release/archive +location with a stable URL, manifest and SHA-256 checksums. Preserve access needed +by reviewers; document retention and the source's redistribution rights. PR +screenshots should be direct GitHub attachments. Do not add new screenshot folders +to `docs/static/img` for evaluation runs. + +Minimize personal data: use reviewed public places and synthetic navigation traces +where possible. Redact private locations, identifiers and credentials before +sharing. Provider photos, ratings and upstream payloads can have separate storage +or redistribution conditions; do not assume the repository's code license covers +them. Retain required source attribution. When a payload cannot be distributed, +record that limitation and permitted derived evidence instead. + +## Sample comparison and negative control + +An initial offline run of this catalog produced 99 passing cases and seven +unavailable manual/device/offline cases, with no guarded known gaps. A second run +on unchanged inputs produced no regressions and no changed fingerprints. The +working tree was dirty during development, so neither was an application-only +comparison. These counts describe this corpus, not the number of supported +product features. + +As a negative control, temporarily remove the autocomplete and aggregate records +from `aachen-hbf-alone.json` in an isolated checkout, retaining its independently +expected station. The search assertion must fail; the comparison must list +`search/aachen-hbf-alone` as a regression and the fixture as a changed input. +Restore the exact original bytes and rerun. This proves the harness notices lost +results without attributing a dataset change to application code. Keep negative +controls temporary; do not weaken expectations or commit the corrupted fixture. + +Read report results alongside their layers and limitations. A provider/data +change, normalization change, presentation change and installed runtime change +require different evidence, even when the same place is involved. diff --git a/docs/docs/developer/map-comparison-baseline.md b/docs/docs/developer/map-comparison-baseline.md index 07ace9a7d..129243d98 100644 --- a/docs/docs/developer/map-comparison-baseline.md +++ b/docs/docs/developer/map-comparison-baseline.md @@ -199,3 +199,8 @@ rank boundaries, exclusions, duplicate prevention, layer priority, and validity. Repeat generation to check idempotence and repeat the visual matrix before expanding the thresholds. Hosted complete MapTiler styles are outside this owned-style policy. + +For versioned search, identity, enrichment and navigation evidence alongside these +visual cases, follow [Discovery evaluation](discovery-evaluation.md). Store new +capture archives externally with a manifest and checksums; attach selected review +images directly to the PR. diff --git a/scripts/discovery-eval/catalog.test.ts b/scripts/discovery-eval/catalog.test.ts new file mode 100644 index 000000000..0a7dd2134 --- /dev/null +++ b/scripts/discovery-eval/catalog.test.ts @@ -0,0 +1,27 @@ +import { readFileSync } from "node:fs"; +import { resolve } from "node:path"; +import { describe, expect, it } from "vitest"; +import { CATALOG, INPUT_FILES } from "./catalog.js"; + +describe("discovery input inventory", () => { + it("fingerprints the recorded upstream station results used by cold/warm ranking", () => { + const entry = CATALOG.find((entry) => entry.id === "search/station-synonyms-cache-order"); + expect(entry?.fixtures).toContain( + "integrations/geocoding-maptiler/__fixtures__/station-search.json", + ); + expect(INPUT_FILES).toContain( + "integrations/geocoding-maptiler/__fixtures__/station-search.json", + ); + }); + + it("includes external JSON imports from every selected suite", () => { + for (const entry of CATALOG.filter((entry) => entry.suite)) { + const suite = entry.suite!; + const source = readFileSync(suite, "utf8"); + for (const match of source.matchAll(/from\s+["'](\.[^"']+\.json)["']/g)) { + const path = resolve(suite, "..", match[1]).slice(`${process.cwd()}/`.length); + expect(INPUT_FILES, `${suite} reads ${path}`).toContain(path); + } + } + }); +}); diff --git a/scripts/discovery-eval/catalog.ts b/scripts/discovery-eval/catalog.ts index 8f92cc888..349b06da9 100644 --- a/scripts/discovery-eval/catalog.ts +++ b/scripts/discovery-eval/catalog.ts @@ -29,6 +29,7 @@ export const CATALOG: EvalCase[] = [ id: "search/station-synonyms-cache-order", layer: "adapted-api/mocked-upstream", suite: "integrations/geocoding/__tests__/forward-ranking.test.ts", + fixtures: ["integrations/geocoding-maptiler/__fixtures__/station-search.json"], note: "Cold/warm query order, station aliases, language/proximity isolation and provider-order fallback; mocked upstream, not provider coverage.", }, { diff --git a/scripts/discovery-eval/report.test.ts b/scripts/discovery-eval/report.test.ts index fbb239aa9..a403c5605 100644 --- a/scripts/discovery-eval/report.test.ts +++ b/scripts/discovery-eval/report.test.ts @@ -94,6 +94,29 @@ describe("versioned discovery evidence reports", () => { }); }); + it("rejects omitted assertions even when a shortened suite passes", () => { + const evidence = input(); + evidence.testResults[0].assertionResults.push({ + fullName: "GPS recovery", + status: "passed", + failureMessages: [], + }); + const wholeSuite = [{ id: "navigation", layer: "synthetic", suite: "search.test.ts" }]; + const before = createReport(evidence, wholeSuite, metadata, "/repo"); + const after = createReport(input(), wholeSuite, metadata, "/repo"); + expect(() => compareReports(before, after)).toThrow("assertion inventory"); + }); + + it.each([ + { passed: 0, failed: 7, unavailable: 0 }, + { passed: 0, failed: 0, unavailable: 0 }, + { passed: 1, failed: 0, unavailable: 1 }, + ])("rejects contradictory passing counts %j", (counts) => { + const report = createReport(input(), cases, metadata, "/repo"); + const broken = { ...report, cases: [{ ...report.cases[0], ...counts }, report.cases[1]] }; + expect(() => compareReports(broken, report)).toThrow("report"); + }); + it("rejects malformed baseline reports and mismatched case sets", () => { const report = createReport(input(), cases, metadata, "/repo"); expect(() => compareReports({} as typeof report, report)).toThrow("report"); diff --git a/scripts/discovery-eval/report.ts b/scripts/discovery-eval/report.ts index 52bf226c8..3c27b0649 100644 --- a/scripts/discovery-eval/report.ts +++ b/scripts/discovery-eval/report.ts @@ -31,6 +31,7 @@ export interface EvalReport { failed: number; unavailable: number; reason?: string; + assertionInventory: string[]; }>; } interface Assertion { @@ -87,7 +88,14 @@ export function createReport( throw new Error("Duplicate evaluation catalog ID"); } const cases = catalog.map((entry): EvalReport["cases"][number] => { - const base = { id: entry.id, layer: entry.layer, passed: 0, failed: 0, unavailable: 0 }; + const base = { + id: entry.id, + layer: entry.layer, + passed: 0, + failed: 0, + unavailable: 0, + assertionInventory: [] as string[], + }; if (!entry.suite) return { ...base, @@ -122,6 +130,10 @@ export function createReport( : "passed"; return { ...base, + // Names can contain test parameters; retain identity without saving their contents. + assertionInventory: selected + .map((assertion) => createHash("sha256").update(assertion.fullName).digest("hex")) + .sort(), status, passed, failed, @@ -176,7 +188,15 @@ function validReport(value: unknown): value is EvalReport { [entry.passed, entry.failed, entry.unavailable].every( (n) => Number.isInteger(n) && (n as number) >= 0, ) && - (entry.reason === undefined || typeof entry.reason === "string"), + Array.isArray(entry.assertionInventory) && + entry.assertionInventory.every( + (id) => typeof id === "string" && /^[a-f0-9]{64}$/.test(id), + ) && + entry.assertionInventory.length === + (entry.passed as number) + (entry.failed as number) + (entry.unavailable as number) && + (entry.reason === undefined || typeof entry.reason === "string") && + (!["passed", "known-gap"].includes(String(entry.status)) || + ((entry.passed as number) > 0 && entry.failed === 0 && entry.unavailable === 0)), ) && new Set(value.cases.map((entry) => entry.id)).size === value.cases.length ); } @@ -200,6 +220,14 @@ export function compareReports(before: unknown, after: unknown) { ) throw new Error("Incompatible evaluation case set"); const previous = new Map(before.cases.map((entry) => [entry.id, entry])); + if ( + after.cases.some( + (entry) => + JSON.stringify(entry.assertionInventory) !== + JSON.stringify(previous.get(entry.id)?.assertionInventory), + ) + ) + throw new Error("Incompatible evaluation assertion inventory; required evidence changed"); const keys = new Set([ ...Object.keys(before.metadata.inputHashes), ...Object.keys(after.metadata.inputHashes), From af1aff37b3067b3830b8731aa497876f14e5146a Mon Sep 17 00:00:00 2001 From: Florian Date: Wed, 7 Oct 2026 21:10:33 +0200 Subject: [PATCH 4/4] feat(eval): complete independently reviewed discovery pilot --- docs/docs/developer/discovery-evaluation.md | 193 ++++-- scripts/discovery-eval/catalog.test.ts | 8 + scripts/discovery-eval/catalog.ts | 33 +- .../examples/control-after.json | 124 ++++ .../examples/control-before.json | 124 ++++ scripts/discovery-eval/options.test.ts | 15 + scripts/discovery-eval/options.ts | 22 + scripts/discovery-eval/report.test.ts | 50 +- scripts/discovery-eval/report.ts | 33 +- scripts/discovery-eval/reviewed.test.ts | 326 ++++++++++ scripts/discovery-eval/reviewed.ts | 576 ++++++++++++++++++ scripts/discovery-eval/run.ts | 42 +- 12 files changed, 1467 insertions(+), 79 deletions(-) create mode 100644 scripts/discovery-eval/examples/control-after.json create mode 100644 scripts/discovery-eval/examples/control-before.json create mode 100644 scripts/discovery-eval/options.test.ts create mode 100644 scripts/discovery-eval/options.ts create mode 100644 scripts/discovery-eval/reviewed.test.ts create mode 100644 scripts/discovery-eval/reviewed.ts diff --git a/docs/docs/developer/discovery-evaluation.md b/docs/docs/developer/discovery-evaluation.md index 53c711bb9..46d2cc813 100644 --- a/docs/docs/developer/discovery-evaluation.md +++ b/docs/docs/developer/discovery-evaluation.md @@ -7,7 +7,7 @@ description: Versioned offline discovery and navigation evidence, with a protoco Use this protocol before changing discovery, place enrichment or navigation. It connects existing semantic search fixtures, conflation guards, Overture gates, -place-card contracts and navigation replays. It does not turn synthetic tests +place-card and selected-sheet contracts and navigation replays. It does not turn synthetic tests into evidence of current regional coverage or installed-device readiness. ## Run and compare @@ -28,12 +28,12 @@ flag, SHA-256 fingerprints of input fixtures, assertion suites, expectations and both map styles. Temporary Vitest evidence is removed after reporting; its error stacks and arbitrary test payloads are excluded from the saved report. -Protocol version 1 rejects a baseline with another version, catalog, budgets or +Protocol version 2 rejects a baseline with another version, catalog, budgets or case set or assertion inventory. Assertion names are retained as hashes, so a shortened passing suite cannot silently replace complete baseline evidence. Changed input fingerprints are listed separately from behavioral -regressions. `appOnlyComparison` requires unchanged fingerprints and two clean -working trees; it establishes comparability, not causality. Commit intentional +regressions. `appOnlyComparison` requires unchanged fingerprints, unchanged observation +conditions/provider inputs, no supplied operator observations and two clean working trees; it establishes comparability, not causality. Commit intentional changes before capturing review evidence. A local experiment in a dirty tree is still useful, but its HEAD does not identify all code that ran. @@ -41,7 +41,10 @@ The command invokes the existing suites without live provider requests. Those suites and the report/capture regressions run through ordinary `pnpm test` in CI; no paid API or deployed dataset is required. Exit status is nonzero for failed runs, missing/skipped required automated assertions or a comparison regression. -Manual/unimplemented cases remain unavailable and do not block this offline gate. +Absent manual/unimplemented cases remain unavailable and do not block this offline +gate. Supplied reviewed observations are scored: failed budgets or lost previously +passing evidence produce a nonzero exit status. A live coverage failure is a useful +measurement, not evidence that the evaluation tool is broken. A failure elsewhere in a selected suite still fails the run, even if individual catalog cases passed. @@ -51,21 +54,22 @@ The versioned catalog is `scripts/discovery-eval/catalog.ts`. Its automated expectations reuse existing tests, rather than scoring the current provider's ordering as truth. -| Case family | Current evidence | Required live complement | -| ------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------- | -| Stations, aliases, multilingual names, category queries and chains | Recorded adapted API inputs; independently specified expected labels, types, IDs and coordinate radii in `search-eval/cases.ts` | Verify the desired entity and each branch against an independent source before re-recording | -| Station synonyms and caches | Mocked upstream cold/warm and query-order regressions; language/proximity partitions | Repeat both query orders on an isolated local cache with a fixed provider configuration | -| Co-located tenants and branches | Synthetic conflation guards for plural-per-address categories, contradictory contacts/addresses and shared switchboards | Judge actual tenants, entrances and floors; equal address does not establish identity | -| Partial place cards | Contract fixtures for photo/rating failures, independent credits, loading, bounded retries and stale searches | Inspect missing facts and delayed requests in the browser with a pinned API revision | -| Urban/rural categories | Unit contracts for the reviewed Overture anchor gate in Aachen, Berlin, Monschau and Maastricht | Run the staged-release gate against an exact imported generation and repeat the visual baseline | -| Closed/missing businesses | Manual case, unavailable in this command | Independently dated closure/existence judgments; filtering code alone is insufficient | -| Transit transitions, recovery and GPS gaps | Synthetic shared-engine replays, including transfers and serialized recovery | Installed shell, permissions, lifecycle and actual-device evidence under #398 | -| Ground off-route behavior, arrival and alternatives | Synthetic shared-engine assertions | Repeat a real or independently recorded route with a pinned routing dataset | -| Offline place search and rerouting | Unavailable pending #403 and #404 | Network-denial or airplane-mode evidence once those capabilities exist | +| Case family | Current evidence | Required live complement | +| ------------------------------------------------------------------ | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------- | +| Stations, aliases, multilingual names, category queries and chains | Recorded adapted API inputs; independently specified expected labels, types, IDs and coordinate radii in `search-eval/cases.ts` | Verify the desired entity and each branch against an independent source before re-recording | +| Station synonyms and caches | Mocked upstream cold/warm and query-order regressions; language/proximity partitions | Repeat both query orders on an isolated local cache with a fixed provider configuration | +| Co-located tenants and branches | Synthetic conflation guards for plural-per-address categories, contradictory contacts/addresses and shared switchboards | Judge actual tenants, entrances and floors; equal address does not establish identity | +| Partial place cards and sheets | Card and PlaceDetailContent contracts for photo/rating failures, absent/uncertain hours, photo detents, independent actions, loading, bounded retries and retained selection | Inspect missing facts and delayed requests in the browser with a pinned API revision | +| Urban/rural categories | Unit contracts for the reviewed Overture anchor gate in Aachen, Berlin, Monschau and Maastricht | Run the staged-release gate against an exact imported generation and repeat the visual baseline | +| Closed/missing businesses | Dated SEA LIFE closure and independently verified EDEKA tenant judgments; operator observations scored when supplied | Independently dated closure/existence judgments; filtering code alone is insufficient | +| Transit transitions, recovery and GPS gaps | Synthetic shared-engine replays, including transfers and serialized recovery | Installed shell, permissions, lifecycle and actual-device evidence under #398 | +| Ground off-route behavior, arrival and alternatives | Synthetic shared-engine assertions | Repeat a real or independently recorded route with a pinned routing dataset | +| Offline place search and rerouting | Unavailable pending #403 and #404 | Network-denial or airplane-mode evidence once those capabilities exist | An assertion passes, fails or is unavailable. An explicitly guarded known gap is reported as `known-gap`, rather than claiming its semantic expectation passed. -Manual cases describe the evidence still needed. Assertion counts may overlap +Reviewed observations list each absent evidence layer; installed/offline cases +describe the evidence still needed. Assertion counts may overlap between cases and are not unique observations, recall values or production performance measurements. @@ -169,23 +173,138 @@ or redistribution conditions; do not assume the repository's code license covers them. Retain required source attribution. When a payload cannot be distributed, record that limitation and permitted derived evidence instead. -## Sample comparison and negative control - -An initial offline run of this catalog produced 99 passing cases and seven -unavailable manual/device/offline cases, with no guarded known gaps. A second run -on unchanged inputs produced no regressions and no changed fingerprints. The -working tree was dirty during development, so neither was an application-only -comparison. These counts describe this corpus, not the number of supported -product features. - -As a negative control, temporarily remove the autocomplete and aggregate records -from `aachen-hbf-alone.json` in an isolated checkout, retaining its independently -expected station. The search assertion must fail; the comparison must list -`search/aachen-hbf-alone` as a regression and the fixture as a changed input. -Restore the exact original bytes and rerun. This proves the harness notices lost -results without attributing a dataset change to application code. Keep negative -controls temporary; do not weaken expectations or commit the corrupted fixture. - -Read report results alongside their layers and limitations. A provider/data -change, normalization change, presentation change and installed runtime change -require different evidence, even when the same place is involved. +## Reviewed pilot cases and operator manifest + +`reviewed.ts` adds nine versioned cases with dated assessor judgments and public +source links: an exact REWE branch, MediaMarkt and EDEKA tenants in ALEXA, a joint +mall tenant inventory, permanently closed SEA LIFE Berlin, and zoom-15 landmarks +at the four existing urban/rural baseline cameras. Retailer websites establish +business identity independently of search ordering; OSM element versions locate +branches/landmarks. Mall coordinates are approximate, with explicit tolerances; +this pilot makes no entrance/floor claim. SEA LIFE's own profile establishes the +closure date. Recheck dated judgments before updating this corpus. A historical +listing with explicit closure context is allowed; an unqualified listing fails. +The mall inventory has no geocoder query: inspect the tenants together rather +than treating an invented query as a provider capability. + +The pilot freezes these rules before comparison: + +| Metric | Pilot rule | +| ------------------------------------ | -------------------------------------------------------------------------------------------------------- | +| Expected search entities | All required entities within the first three results and their coordinate radii | +| Selected wrong branch | Zero wrong-brand-branch selections in the first result; later alternative branches are allowed | +| Duplicates / unqualified closed hits | Zero duplicates of an expected entity / zero excluded-entity hits without explicit closure context | +| Browsing | Expected landmark anywhere in the fully readable set; at least one useful label; zero overlapping labels | +| Requests / latency | At most one request per captured operation; cold ≤5,000 ms, warm ≤2,000 ms | + +These are small-pilot acceptance rules, not universal SLAs or population-wide +coverage thresholds. Uncontrolled-cache timings are retained but do not pass/fail +a cold/warm latency budget. Controlled cold/warm observations require declared isolated +cache conditions; a shared/unknown cache must be marked uncontrolled. Unmeasured values are `null`, not zero. A browsing +presentation observation lacking a readable-label or overlap judgment is +unavailable. Rendered feature counts include clipped/obscured labels and cannot +substitute for a visual judgment. Rank is not meaningful for an unordered map +feature set. + +Supply an operator-owned JSON manifest: + +```bash +pnpm discovery-eval --evidence /tmp/operator/manifest.json \ + --out /tmp/operator/report +``` + +Use `scripts/discovery-eval/examples/control-before.json` as the **schema example**, +not live evidence. Paths are resolved from the repository root. The strict schema +requires version, assessor/time, region, deployment/style/extract/source revisions, +provider/capabilities, reviewed-case query order, cache isolation, +language/theme/viewport/DPR and observations. Revisions are +`{ "value": null, "reason": "why unavailable" }` or a known value with a null +reason. The local app commit and owned style/input hashes are recorded separately; +a public runtime style checksum does not identify the deployment commit. A +replication timestamp does not prove the extract date/checksum. Do not copy an +old generation into a new manifest merely because the region is unchanged. + +Each observation identifies its reviewed case, one evidence layer and processing +stage, provenance +(`live`, `recorded`, `synthetic`) and cache condition. Results contain only label, +`[longitude, latitude]` and whether closure context was explicitly shown. Capture +at the case's query/camera/zoom with zero pitch/bearing, and retain order for +search. Record request count, latency in ms, useful readable POI/park labels and +label-overlap count; use `null` for measurements not made. The schema rejects +unknown fields, invalid coordinates/counts and duplicate case/layer observations; +its errors omit rejected input. A single manifest is one capture per case/layer; +use separate manifests for cold/warm or query-order comparisons. + +Layer meanings are fixed: + +- `data` / `source`: independently inspected imported/source records, not inferred from API output. +- `provider` / `raw-upstream`: permitted raw-upstream capture projected onto the judged entities. +- `normalization` / `adapted-api` or `client-ranking`: distinct stages; use + separate reports to inspect both. Unlike stages cannot produce numeric deltas. +- `presentation` / `final-ui`: final readable browser rows/labels, selection and sheet state. +- `runtime` / `engine-replay` or `installed-device`: observed execution, distinct from the automated navigation-engine + replay cases in the main report. + +Do not relabel adapted API output as upstream. Existing recorded client-ranking +cases and mocked adapter/cache cases remain separate in the automated catalog. +The report retains supplied context, definitions, budgets, numerical measurements +and each absent case/layer. Comparison reports list metric changes even when both +runs pass, changed source/configuration or capture conditions, changed data/provider +results, and lost passing evidence. Removing a previously measured value is an evidence +regression even if the remaining entity results pass. Unknown revisions remain unknown; identical +unknowns do not establish an application-only comparison. Operator manifests compare +observed captures; `appOnlyComparison` is reserved for clean offline assertion runs +without supplied observations, whose tested inputs are fingerprinted. + +## Worked comparison and negative control + +Run the committed, deliberately synthetic control fixtures: + +```bash +pnpm discovery-eval --out /tmp/control-before \ + --evidence scripts/discovery-eval/examples/control-before.json +pnpm discovery-eval --out /tmp/control-after \ + --evidence scripts/discovery-eval/examples/control-after.json \ + --baseline /tmp/control-before/report.json +# The second command must exit 1: the selected REWE branch was deliberately moved. +``` + +Coordinate judgments reference [© OpenStreetMap contributors, ODbL](https://www.openstreetmap.org/copyright); +official website references supply independently reviewed public identity facts, +not permission to redistribute their pages or photos. + +The fixtures use independently reviewed entity facts and invented measurements; +they contain no downloaded provider payloads or paid API requests. CI tests this +same comparison. This is a harness demonstration, not an application improvement: + +| Outcome | Before | After | Interpretation | +| ------------- | --------------------------------------------- | ------------------------------------------ | ----------------------------------------------------------------------------- | +| Data | Required branch present | Unchanged | Synthetic source projection unchanged | +| Provider | Correct branch first | Unchanged | Synthetic upstream projection unchanged | +| Normalization | Recall 1, rank 1, wrong branch 0 | Recall 0, rank unavailable, wrong branch 1 | Deliberate normalized-result regression; reported separately | +| Presentation | Landmark recall 1, useful labels 6, overlap 0 | Unchanged | Synthetic readable-label judgment unchanged | +| Runtime | Existing navigation replays pass | Unchanged | Shared-engine synthetic evidence; installed/offline runtime still unavailable | + +The command writes complete `report.json` and `report.md` in both output folders. +The after report has `runSucceeded: false` and the reviewed regression +`business/rewe-invalidenstrasse/normalization`, with no changed data/provider +inputs. A rank-1 → rank-2 change is also detected numerically even when it remains +within the top-three budget. Separate regression tests remove a tenant/business, +reintroduce an unqualified closed attraction, remove previous evidence and reject +tampered baseline metrics. + +An October 7, 2026 read-only live pilot on OpenMapX.com illustrates the distinction: +REWE's exact branch appears; EDEKA's independently verified ALEXA business is +missing from its adapted autocomplete results. Four dark-theme phone viewports +(430×932, DPR 1, English, zoom 15) were visually inspected: Berlin's cathedral +feature exists but its label is clipped; Aachen's cathedral and Monschau's castle +are readable; Neuss's expected minster label is absent. These are observations of +that deployment, not universal judgments. The API cache was uncontrolled; raw +upstream, geocoder extract revisions and deployment commit were unavailable. +Keep the operator manifest/reports/screenshots outside Git, with checksums and +rights/access notes. A public observation is not a clean before/after experiment. + +For navigation or station-search changes, the existing semantic assertions must +still fail when expected events/entities are lost. Do not weaken expectations to +make a changed result pass. Read results alongside their layers: data, provider, +normalization, presentation and installed runtime require different evidence. diff --git a/scripts/discovery-eval/catalog.test.ts b/scripts/discovery-eval/catalog.test.ts index 0a7dd2134..bee7fe2f7 100644 --- a/scripts/discovery-eval/catalog.test.ts +++ b/scripts/discovery-eval/catalog.test.ts @@ -4,6 +4,14 @@ import { describe, expect, it } from "vitest"; import { CATALOG, INPUT_FILES } from "./catalog.js"; describe("discovery input inventory", () => { + it("runs selected-place partial states as well as list-card enrichment", () => { + expect( + CATALOG.some( + (entry) => + entry.suite === "apps/web/src/components/panels/place/PlaceDetailContent.test.tsx", + ), + ).toBe(true); + }); it("fingerprints the recorded upstream station results used by cold/warm ranking", () => { const entry = CATALOG.find((entry) => entry.id === "search/station-synonyms-cache-order"); expect(entry?.fixtures).toContain( diff --git a/scripts/discovery-eval/catalog.ts b/scripts/discovery-eval/catalog.ts index 349b06da9..abc7e0788 100644 --- a/scripts/discovery-eval/catalog.ts +++ b/scripts/discovery-eval/catalog.ts @@ -49,6 +49,12 @@ export const CATALOG: EvalCase[] = [ suite: cards, note: "Partial photos/ratings, independent credits, bounded retries, stale searches and missing data; no real-provider availability or production latency claim.", }, + { + id: "place/selected-sheet-partial-states", + layer: "ui/contract-fixtures", + suite: "apps/web/src/components/panels/place/PlaceDetailContent.test.tsx", + note: "Absent/uncertain hours, photo detents, retained selection across enrichment and independent actions; no availability claim.", + }, { id: "coverage/overture-reviewed-gate-contract", layer: "dataset-gate/unit-fixtures", @@ -85,30 +91,6 @@ export const CATALOG: EvalCase[] = [ layer: "navigation-engine/synthetic", suite: `${navigation}/fasterRoute.test.ts`, }, - { - id: "manual/urban-rural-browsing", - layer: "ui/live-manual", - unavailable: - "Repeat the map comparison baseline at fixed viewport, coordinates, zoom and provider/style/source revisions; attach external capture manifest and judgments.", - }, - { - id: "manual/closed-missing-businesses", - layer: "dataset/live-manual", - unavailable: - "Independently verify a stratified regional sample of missing/closed businesses against exact OSM/Overture generations; no current closure ground truth is captured here.", - }, - { - id: "manual/mall-tenant-floor-identity", - layer: "dataset/live-manual", - unavailable: - "Independently judge tenants, branches, entrances and floors; synthetic conflation guards do not establish regional coverage.", - }, - { - id: "manual/provider-performance", - layer: "adapted-api/live-manual", - unavailable: - "Declare request/latency budgets and repeat cold/warm captures in both query orders, with explicit region/configuration/source revisions.", - }, { id: "unavailable/installed-navigation", layer: "installed-device", @@ -136,6 +118,9 @@ export const INPUT_FILES = [ ...(entry.suite ? [entry.suite] : []), ]), `${search}/cases.ts`, + "scripts/discovery-eval/reviewed.ts", + "scripts/discovery-eval/examples/control-before.json", + "scripts/discovery-eval/examples/control-after.json", "services/data-manager/src/jobs/overture/eval/quality-baseline.ts", "apps/web/public/styles/openmapx-streets.json", "apps/web/public/styles/openmapx-dark.json", diff --git a/scripts/discovery-eval/examples/control-after.json b/scripts/discovery-eval/examples/control-after.json new file mode 100644 index 000000000..2eb0aa725 --- /dev/null +++ b/scripts/discovery-eval/examples/control-after.json @@ -0,0 +1,124 @@ +{ + "version": 1, + "assessedAt": "2026-10-07", + "assessor": "Synthetic control; primary-source case judgments are in reviewed.ts", + "context": { + "region": "Germany: Berlin pilot", + "extractDate": { + "value": null, + "reason": "Synthetic records, no extract used" + }, + "style": { + "value": null, + "reason": "Synthetic display judgment, no rendered style used" + }, + "deployment": { + "value": null, + "reason": "No deployment used in this synthetic example" + }, + "sources": { + "osm": { + "value": null, + "reason": "Coordinates from reviewed case, no imported OSM generation" + }, + "overture": { + "value": null, + "reason": "Not used" + } + }, + "provider": { + "id": "synthetic-control", + "capabilities": ["projected search rows"] + }, + "configuration": { + "language": "en", + "theme": "dark", + "viewport": [430, 932], + "dpr": 1 + }, + "queryOrder": ["business/rewe-invalidenstrasse"], + "cacheIsolation": "isolated" + }, + "observations": [ + { + "caseId": "business/rewe-invalidenstrasse", + "layer": "data", + "kind": "synthetic", + "cache": "cold", + "results": [ + { + "label": "REWE", + "coordinates": [13.3970838, 52.5319807], + "closed": false + } + ], + "measurements": { + "requests": 1, + "latencyMs": 10, + "usefulLabels": null, + "overlaps": null + }, + "stage": "source" + }, + { + "caseId": "business/rewe-invalidenstrasse", + "layer": "provider", + "kind": "synthetic", + "cache": "cold", + "results": [ + { + "label": "REWE", + "coordinates": [13.3970838, 52.5319807], + "closed": false + } + ], + "measurements": { + "requests": 1, + "latencyMs": 10, + "usefulLabels": null, + "overlaps": null + }, + "stage": "raw-upstream" + }, + { + "caseId": "business/rewe-invalidenstrasse", + "layer": "normalization", + "kind": "synthetic", + "cache": "cold", + "results": [ + { + "label": "REWE", + "coordinates": [13.3893156, 52.525331], + "closed": false + } + ], + "measurements": { + "requests": 1, + "latencyMs": 10, + "usefulLabels": null, + "overlaps": null + }, + "stage": "adapted-api" + }, + { + "caseId": "browsing/berlin-z15", + "layer": "presentation", + "kind": "synthetic", + "cache": "uncontrolled", + "results": [ + { + "label": "Berlin Cathedral", + "coordinates": [13.401094, 52.519084], + "closed": false + } + ], + "measurements": { + "requests": null, + "latencyMs": null, + "usefulLabels": 6, + "overlaps": 0 + }, + "stage": "final-ui" + } + ] +} diff --git a/scripts/discovery-eval/examples/control-before.json b/scripts/discovery-eval/examples/control-before.json new file mode 100644 index 000000000..609372dbd --- /dev/null +++ b/scripts/discovery-eval/examples/control-before.json @@ -0,0 +1,124 @@ +{ + "version": 1, + "assessedAt": "2026-10-07", + "assessor": "Synthetic control; primary-source case judgments are in reviewed.ts", + "context": { + "region": "Germany: Berlin pilot", + "extractDate": { + "value": null, + "reason": "Synthetic records, no extract used" + }, + "style": { + "value": null, + "reason": "Synthetic display judgment, no rendered style used" + }, + "deployment": { + "value": null, + "reason": "No deployment used in this synthetic example" + }, + "sources": { + "osm": { + "value": null, + "reason": "Coordinates from reviewed case, no imported OSM generation" + }, + "overture": { + "value": null, + "reason": "Not used" + } + }, + "provider": { + "id": "synthetic-control", + "capabilities": ["projected search rows"] + }, + "configuration": { + "language": "en", + "theme": "dark", + "viewport": [430, 932], + "dpr": 1 + }, + "queryOrder": ["business/rewe-invalidenstrasse"], + "cacheIsolation": "isolated" + }, + "observations": [ + { + "caseId": "business/rewe-invalidenstrasse", + "layer": "data", + "kind": "synthetic", + "cache": "cold", + "results": [ + { + "label": "REWE", + "coordinates": [13.3970838, 52.5319807], + "closed": false + } + ], + "measurements": { + "requests": 1, + "latencyMs": 10, + "usefulLabels": null, + "overlaps": null + }, + "stage": "source" + }, + { + "caseId": "business/rewe-invalidenstrasse", + "layer": "provider", + "kind": "synthetic", + "cache": "cold", + "results": [ + { + "label": "REWE", + "coordinates": [13.3970838, 52.5319807], + "closed": false + } + ], + "measurements": { + "requests": 1, + "latencyMs": 10, + "usefulLabels": null, + "overlaps": null + }, + "stage": "raw-upstream" + }, + { + "caseId": "business/rewe-invalidenstrasse", + "layer": "normalization", + "kind": "synthetic", + "cache": "cold", + "results": [ + { + "label": "REWE", + "coordinates": [13.3970838, 52.5319807], + "closed": false + } + ], + "measurements": { + "requests": 1, + "latencyMs": 10, + "usefulLabels": null, + "overlaps": null + }, + "stage": "adapted-api" + }, + { + "caseId": "browsing/berlin-z15", + "layer": "presentation", + "kind": "synthetic", + "cache": "uncontrolled", + "results": [ + { + "label": "Berlin Cathedral", + "coordinates": [13.401094, 52.519084], + "closed": false + } + ], + "measurements": { + "requests": null, + "latencyMs": null, + "usefulLabels": 6, + "overlaps": 0 + }, + "stage": "final-ui" + } + ] +} diff --git a/scripts/discovery-eval/options.test.ts b/scripts/discovery-eval/options.test.ts new file mode 100644 index 000000000..428b09112 --- /dev/null +++ b/scripts/discovery-eval/options.test.ts @@ -0,0 +1,15 @@ +import { expect, it } from "vitest"; +import { parseOptions } from "./options.js"; + +it("resolves operator paths relative to the repository despite pnpm's CLI workspace", () => { + expect( + parseOptions( + ["--evidence", "scripts/example.json", "--baseline", "before/report.json"], + "/repo", + ), + ).toMatchObject({ evidence: "/repo/scripts/example.json", baseline: "/repo/before/report.json" }); +}); +it("rejects unknown flags and missing paths without repeating potentially secret input", () => { + expect(() => parseOptions(["--api-key", "secret"], "/repo")).toThrow(/^Usage:/); + expect(() => parseOptions(["--evidence"], "/repo")).toThrow(/^Usage:/); +}); diff --git a/scripts/discovery-eval/options.ts b/scripts/discovery-eval/options.ts new file mode 100644 index 000000000..c8aeef476 --- /dev/null +++ b/scripts/discovery-eval/options.ts @@ -0,0 +1,22 @@ +import { join, resolve } from "node:path"; + +export function parseOptions(args: string[], root: string) { + const result = { + out: join(root, ".superpowers/eval-reports/latest"), + baseline: "", + evidence: "", + }; + for (let i = 0; i < args.length; i += 2) { + if ( + !["--out", "--baseline", "--evidence"].includes(args[i]) || + !args[i + 1] || + args[i + 1].startsWith("--") + ) { + throw new Error( + "Usage: pnpm discovery-eval [--out DIRECTORY] [--baseline REPORT.json] [--evidence MANIFEST.json]", + ); + } + result[args[i].slice(2) as keyof typeof result] = resolve(root, args[i + 1]); + } + return result; +} diff --git a/scripts/discovery-eval/report.test.ts b/scripts/discovery-eval/report.test.ts index a403c5605..3542f605d 100644 --- a/scripts/discovery-eval/report.test.ts +++ b/scripts/discovery-eval/report.test.ts @@ -1,5 +1,7 @@ +import { readFileSync } from "node:fs"; import { describe, expect, it } from "vitest"; import { compareReports, createReport, type EvalCase, type EvidenceMetadata } from "./report.js"; +import { buildReviewedEvidence, readManifest } from "./reviewed.js"; const cases: EvalCase[] = [ { id: "station", layer: "client-ranking", suite: "search.test.ts", assertions: ["station rank"] }, @@ -128,7 +130,7 @@ describe("versioned discovery evidence reports", () => { it("rejects comparisons across protocol, catalog or budget changes", () => { const before = createReport(input(), cases, metadata, "/repo"); - expect(() => compareReports(before, { ...before, protocolVersion: 2 })).toThrow("protocol"); + expect(() => compareReports(before, { ...before, protocolVersion: 3 })).toThrow("protocol"); const changed = createReport( input(), [{ ...cases[0], assertions: ["different budget"] }], @@ -144,4 +146,50 @@ describe("versioned discovery evidence reports", () => { expect(() => createReport(evidence, cases, metadata, "/repo")).toThrow("test evidence"); }, ); + it("retains layer-specific unavailable judgments in the generated report", () => { + const report = createReport(input(), cases, metadata, "/repo"); + expect(report.reviewed.unavailable).toContainEqual({ + caseId: "business/edeka-alexa", + layer: "provider", + }); + expect( + report.reviewed.definitions.find((entry) => entry.id === "business/edeka-alexa")?.entities[0] + .label, + ).toBe("EDEKA"); + }); + + it("rejects tampered numerical evidence in a baseline", () => { + const report = createReport(input(), cases, metadata, "/repo"); + const changed = { + ...report, + reviewed: { ...buildReviewedEvidence(null), results: [{ status: "passed" }] }, + }; + expect(() => compareReports(changed, report)).toThrow("report"); + }); + it("rejects array-valued status fields in supplied test evidence and baselines", () => { + const evidence = input(); + expect(() => + createReport( + { ...evidence, testResults: [{ ...evidence.testResults[0], status: ["passed"] }] }, + cases, + metadata, + "/repo", + ), + ).toThrow("test evidence"); + const report = createReport(input(), cases, metadata, "/repo"); + expect(() => + compareReports( + { ...report, cases: [{ ...report.cases[0], status: ["passed"] }, report.cases[1]] }, + report, + ), + ).toThrow("report"); + }); + it("does not infer an application-only change from external recorded observations with unknown deployment", () => { + const manifest = readManifest( + JSON.parse(readFileSync(new URL("./examples/control-before.json", import.meta.url), "utf8")), + ); + for (const entry of manifest.observations) entry.kind = "recorded"; + const report = createReport(input(), cases, metadata, "/repo", manifest); + expect(compareReports(report, report).appOnlyComparison).toBe(false); + }); }); diff --git a/scripts/discovery-eval/report.ts b/scripts/discovery-eval/report.ts index 3c27b0649..f84d982b5 100644 --- a/scripts/discovery-eval/report.ts +++ b/scripts/discovery-eval/report.ts @@ -1,5 +1,13 @@ import { createHash } from "node:crypto"; import { relative } from "node:path"; +import { + buildReviewedEvidence, + compareReviewedEvidence, + type Manifest, + type ReviewedEvidence, + readManifest, + validReviewedEvidence, +} from "./reviewed.js"; export interface EvalCase { id: string; @@ -23,6 +31,7 @@ export interface EvalReport { catalogHash: string; runSucceeded: boolean; metadata: EvidenceMetadata; + reviewed: ReviewedEvidence; cases: Array<{ id: string; layer: string; @@ -55,7 +64,8 @@ function testEvidence(input: unknown): { success: boolean; testResults: TestFile if ( !record(file) || typeof file.name !== "string" || - !["passed", "failed"].includes(String(file.status)) || + typeof file.status !== "string" || + !["passed", "failed"].includes(file.status) || !Array.isArray(file.assertionResults) ) { throw new Error("Invalid test evidence"); @@ -82,6 +92,7 @@ export function createReport( catalog: EvalCase[], metadata: EvidenceMetadata, root: string, + manifest: Manifest | null = null, ): EvalReport { const evidence = testEvidence(input); if (new Set(catalog.map((entry) => entry.id)).size !== catalog.length) { @@ -146,11 +157,14 @@ export function createReport( : {}), }; }); + const reviewed = buildReviewedEvidence(manifest ? readManifest(manifest) : null); return { - protocolVersion: 1, + protocolVersion: 2, + reviewed, catalogHash: createHash("sha256").update(JSON.stringify(catalog)).digest("hex"), runSucceeded: evidence.success && + reviewed.results.every((entry) => entry.status !== "failed") && cases.every( (entry, index) => !catalog[index].suite || ["passed", "known-gap"].includes(entry.status), ), @@ -162,7 +176,7 @@ export function createReport( function validReport(value: unknown): value is EvalReport { if ( !record(value) || - value.protocolVersion !== 1 || + value.protocolVersion !== 2 || typeof value.catalogHash !== "string" || typeof value.runSucceeded !== "boolean" || !record(value.metadata) || @@ -175,6 +189,7 @@ function validReport(value: unknown): value is EvalReport { value.metadata.captureProvenance.legacyFixtures, value.metadata.captureProvenance.adaptedApiCaptures, ].every((n) => Number.isInteger(n) && (n as number) >= 0) || + !validReviewedEvidence(value.reviewed) || !Array.isArray(value.cases) ) return false; @@ -184,7 +199,8 @@ function validReport(value: unknown): value is EvalReport { record(entry) && typeof entry.id === "string" && typeof entry.layer === "string" && - ["passed", "failed", "unavailable", "known-gap"].includes(String(entry.status)) && + typeof entry.status === "string" && + ["passed", "failed", "unavailable", "known-gap"].includes(entry.status) && [entry.passed, entry.failed, entry.unavailable].every( (n) => Number.isInteger(n) && (n as number) >= 0, ) && @@ -235,7 +251,12 @@ export function compareReports(before: unknown, after: unknown) { const changedInputs = [...keys] .filter((key) => before.metadata.inputHashes[key] !== after.metadata.inputHashes[key]) .sort(); + const reviewed = compareReviewedEvidence(before.reviewed, after.reviewed); + const hasOperatorObservations = [before, after].some( + (report) => (report.reviewed.manifest?.observations.length ?? 0) > 0, + ); return { + reviewed, regressions: after.cases .filter((entry) => previous.get(entry.id)?.status === "passed" && entry.status !== "passed") .map((entry) => entry.id), @@ -245,6 +266,10 @@ export function compareReports(before: unknown, after: unknown) { changedInputs, appOnlyComparison: changedInputs.length === 0 && + !hasOperatorObservations && + !reviewed.contextChanged && + !reviewed.captureConditionsChanged && + !reviewed.changedProviderInputs.length && !before.metadata.workingTreeDirty && !after.metadata.workingTreeDirty, runRegression: before.runSucceeded && !after.runSucceeded, diff --git a/scripts/discovery-eval/reviewed.test.ts b/scripts/discovery-eval/reviewed.test.ts new file mode 100644 index 000000000..8d434858d --- /dev/null +++ b/scripts/discovery-eval/reviewed.test.ts @@ -0,0 +1,326 @@ +import { readFileSync } from "node:fs"; +import { describe, expect, it } from "vitest"; +import { + assessObservation, + buildReviewedEvidence, + compareAssessments, + compareReviewedEvidence, + type Manifest, + REVIEWED_CASES, + readManifest, + validReviewedEvidence, +} from "./reviewed.js"; + +const branch = () => REVIEWED_CASES.find((entry) => entry.id === "business/rewe-invalidenstrasse")!; +const observed = (label: string, coordinates: [number, number]) => ({ + label, + coordinates, + closed: false, +}); +const context: Manifest["context"] = { + queryOrder: ["business/rewe-invalidenstrasse"], + cacheIsolation: "unknown", + region: "Germany: Berlin, Aachen, Neuss, Monschau", + extractDate: { value: null, reason: "Exact deployed extract dates are not exposed" }, + style: { value: null, reason: "Not captured in this API-only observation" }, + deployment: { value: null, reason: "Public deployment does not expose its commit" }, + sources: { + osm: { value: null, reason: "No extract checksum exposed" }, + overture: { value: null, reason: "No deployed generation exposed" }, + }, + provider: { id: "geocoding-maptiler", capabilities: ["autocomplete"] }, + configuration: { language: "en", theme: "dark", viewport: [430, 932], dpr: 1 }, +}; +const manifest = (): Manifest => ({ + version: 1, + assessedAt: "2026-10-07T18:00:00Z", + assessor: "Codex assisted review", + context, + observations: [ + { + caseId: "business/rewe-invalidenstrasse", + layer: "provider", + stage: "raw-upstream", + kind: "live", + cache: "uncontrolled", + results: [observed("REWE", [13.3970838, 52.5319807])], + measurements: { requests: 1, latencyMs: 500, usefulLabels: null, overlaps: null }, + }, + ], +}); + +describe("independently reviewed discovery observations", () => { + it("rejects a different same-brand branch instead of accepting its name", () => { + const result = assessObservation(branch(), { + ...manifest().observations[0], + results: [observed("REWE", [13.3893156, 52.525331])], + }); + expect(result.metrics).toMatchObject({ recall: 0, firstRank: null, wrongBranch: 1 }); + expect(result.status).toBe("failed"); + }); + + it("retains two independently expected tenants at the same mall address", () => { + const entry = REVIEWED_CASES.find((entry) => entry.id === "business/alexa-tenants")!; + const observation = { + ...manifest().observations[0], + caseId: entry.id, + results: [observed("MediaMarkt", [13.41479, 52.51986])], + }; + expect(assessObservation(entry, observation).metrics.recall).toBe(0.5); + expect( + assessObservation(entry, { + ...observation, + results: [...observation.results, observed("EDEKA Moch", [13.41479, 52.51986])], + }).status, + ).toBe("passed"); + }); + + it("counts a missing independently verified business, not its street address, as a miss", () => { + const entry = REVIEWED_CASES.find((entry) => entry.id === "business/edeka-alexa")!; + expect( + assessObservation(entry, { + ...manifest().observations[0], + caseId: entry.id, + results: [observed("Grunerstraße", [13.41479, 52.51986])], + }).metrics.recall, + ).toBe(0); + }); + + it("detects reappearance of a dated permanently closed business", () => { + const entry = REVIEWED_CASES.find((entry) => entry.id === "business/sealife-closed")!; + const observation = { ...manifest().observations[0], caseId: entry.id, results: [] }; + expect(assessObservation(entry, observation).status).toBe("passed"); + expect( + assessObservation(entry, { + ...observation, + results: [observed("SEA LIFE Berlin", [13.4028, 52.5203])], + }).metrics.forbiddenHits, + ).toBe(1); + expect( + assessObservation(entry, { + ...observation, + results: [observed("SEA LIFE Berlin", [13.4028, 52.5203])], + }).status, + ).toBe("failed"); + }); + + it("keeps uncontrolled-cache timing measurable without claiming it is a warm-cache result", () => { + const result = assessObservation(branch(), manifest().observations[0]); + expect(result.metrics.latencyMs).toBe(500); + expect(result.cache).toBe("uncontrolled"); + expect( + assessObservation(branch(), { + ...manifest().observations[0], + cache: "warm", + measurements: { requests: 1, latencyMs: 60000, usefulLabels: null, overlaps: null }, + }).status, + ).toBe("failed"); + }); + + it("requires explicit reasons for unknown revisions and rejects arbitrary configuration", () => { + expect(readManifest(manifest()).context.deployment.value).toBeNull(); + expect(() => + readManifest({ ...manifest(), context: { ...context, deployment: { value: null } } }), + ).toThrow("manifest"); + expect(() => + readManifest({ + ...manifest(), + context: { ...context, configuration: { ...context.configuration, apiKey: "secret" } }, + }), + ).toThrow("manifest"); + }); + + it("rejects duplicate case/layer observations and unrecognized cases", () => { + const value = manifest(); + expect(() => + readManifest({ ...value, observations: [...value.observations, ...value.observations] }), + ).toThrow("manifest"); + expect(() => + readManifest({ ...value, observations: [{ ...value.observations[0], caseId: "invented" }] }), + ).toThrow("manifest"); + }); + + it("reports numerical rank deterioration even while both results remain within budget", () => { + const value = readManifest(manifest()); + const before = assessObservation(branch(), value.observations[0]); + const after = assessObservation(branch(), { + ...value.observations[0], + results: [ + observed("Some unrelated business", [13.4, 52.52]), + ...value.observations[0].results, + ], + }); + expect(compareAssessments([before], [after]).changes).toContainEqual({ + caseId: "business/rewe-invalidenstrasse", + layer: "provider", + metric: "firstRank", + before: 1, + after: 2, + }); + }); + + it("retains explicit unavailable layers rather than inventing upstream or installed runtime evidence", () => { + const result = buildReviewedEvidence(readManifest(manifest())); + expect(result.results[0].metrics.firstRank).toBe(1); + expect(result.unavailable).toContainEqual({ + caseId: "business/rewe-invalidenstrasse", + layer: "data", + }); + expect(result.unavailable).toContainEqual({ + caseId: "business/rewe-invalidenstrasse", + layer: "runtime", + }); + }); + + it("separates changed source configuration and changed provider results from application-only evidence", () => { + const before = buildReviewedEvidence(readManifest(manifest())); + const changed = manifest(); + changed.context = { + ...context, + sources: { ...context.sources, osm: { value: "OSM extract 2026-10-06", reason: null } }, + }; + const after = buildReviewedEvidence(readManifest(changed)); + expect(compareReviewedEvidence(before, after).contextChanged).toBe(true); + const missing = manifest(); + missing.observations[0].results = []; + expect( + compareReviewedEvidence(before, buildReviewedEvidence(readManifest(missing))) + .changedProviderInputs, + ).toContain("business/rewe-invalidenstrasse/provider"); + }); + + it("detects removal of previously present evidence", () => { + const before = buildReviewedEvidence(readManifest(manifest())); + const after = buildReviewedEvidence(readManifest({ ...manifest(), observations: [] })); + expect(compareReviewedEvidence(before, after).regressions).toContain( + "business/rewe-invalidenstrasse/provider", + ); + }); + it("accepts explicitly closed historical listings without treating them as operating businesses", () => { + const entry = REVIEWED_CASES.find((entry) => entry.id === "business/sealife-closed")!; + expect( + assessObservation(entry, { + ...manifest().observations[0], + caseId: entry.id, + results: [{ ...observed("SEA LIFE Berlin", [13.4028, 52.5203]), closed: true }], + }).status, + ).toBe("passed"); + }); + + it("does not fail a correct selected branch because later alternatives include other branches", () => { + expect( + assessObservation(branch(), { + ...manifest().observations[0], + results: [ + observed("REWE", [13.3970838, 52.5319807]), + observed("REWE", [13.3893156, 52.525331]), + ], + }).status, + ).toBe("passed"); + }); + + it("scores a browsing landmark anywhere in the visible set, not by arbitrary feature order", () => { + const entry = REVIEWED_CASES.find((entry) => entry.id === "browsing/berlin-z15")!; + const observation = { + ...manifest().observations[0], + caseId: entry.id, + layer: "presentation" as const, + stage: "final-ui" as const, + results: [ + ...Array.from({ length: 5 }, () => observed("Unrelated", [13.4, 52.52])), + observed("Berlin Cathedral", [13.401094, 52.519084]), + ], + measurements: { requests: null, latencyMs: null, usefulLabels: 6, overlaps: 0 }, + }; + expect(assessObservation(entry, observation).status).toBe("passed"); + }); + + it("rejects mismatched observations and fractional count measurements", () => { + expect(() => + assessObservation(branch(), { + ...manifest().observations[0], + caseId: "business/edeka-alexa", + }), + ).toThrow("case"); + const value = manifest(); + value.observations[0].measurements.requests = 0.5; + expect(() => readManifest(value)).toThrow("manifest"); + }); + + it("validates evidence independently of JSON property order and detects tampered metrics", () => { + const evidence = buildReviewedEvidence(readManifest(manifest())); + expect(validReviewedEvidence(Object.fromEntries(Object.entries(evidence).reverse()))).toBe( + true, + ); + evidence.results[0].metrics.recall = 0; + expect(validReviewedEvidence(evidence)).toBe(false); + }); + it("detects the checked-in control's normalization regression while data/provider/presentation stay unchanged", () => { + const load = (name: string) => + buildReviewedEvidence( + readManifest( + JSON.parse( + readFileSync(new URL(`./examples/control-${name}.json`, import.meta.url), "utf8"), + ), + ), + ); + const before = load("before"), + after = load("after"); + const result = compareReviewedEvidence(before, after); + expect(result.regressions).toEqual(["business/rewe-invalidenstrasse/normalization"]); + expect(result.changedProviderInputs).toEqual([]); + expect(result.changes.filter((entry) => entry.layer !== "normalization")).toEqual([]); + }); + it("rejects array-valued enums instead of coercing them into trusted capture conditions", () => { + for (const key of ["kind", "cache"] as const) { + const value = manifest(); + expect(() => + readManifest({ + ...value, + observations: [{ ...value.observations[0], [key]: [value.observations[0][key]] }], + }), + ).toThrow("manifest"); + } + expect(() => + readManifest({ + ...manifest(), + context: { ...context, configuration: { ...context.configuration, theme: ["dark"] } }, + }), + ).toThrow("manifest"); + }); + + it("records query order/cache isolation and refuses unlike processing stages as a numerical comparison", () => { + const before = buildReviewedEvidence(readManifest(manifest())); + const value = manifest(); + value.observations[0].layer = "normalization"; + value.observations[0].stage = "adapted-api"; + const first = buildReviewedEvidence(readManifest(value)); + value.observations[0].stage = "client-ranking"; + const next = buildReviewedEvidence(readManifest(value)); + expect(compareReviewedEvidence(first, next).captureConditionsChanged).toBe(true); + expect(compareReviewedEvidence(first, next).changes).toEqual([]); + expect(compareReviewedEvidence(first, next).regressions).toContain( + "business/rewe-invalidenstrasse/normalization", + ); + expect(before.manifest?.context.queryOrder).toEqual(["business/rewe-invalidenstrasse"]); + expect(() => + readManifest({ + ...manifest(), + context: { ...context, cacheIsolation: "shared" }, + observations: [{ ...manifest().observations[0], cache: "cold" }], + }), + ).toThrow("manifest"); + }); + + it("treats lost previously measured performance evidence as a regression", () => { + const before = buildReviewedEvidence(readManifest(manifest())); + const value = manifest(); + value.observations[0].measurements.requests = null; + value.observations[0].measurements.latencyMs = null; + const after = buildReviewedEvidence(readManifest(value)); + expect(after.results[0].status).toBe("passed"); + expect(compareReviewedEvidence(before, after).regressions).toContain( + "business/rewe-invalidenstrasse/provider", + ); + }); +}); diff --git a/scripts/discovery-eval/reviewed.ts b/scripts/discovery-eval/reviewed.ts new file mode 100644 index 000000000..7a89f971e --- /dev/null +++ b/scripts/discovery-eval/reviewed.ts @@ -0,0 +1,576 @@ +import { createHash } from "node:crypto"; +import { haversineDistance } from "../../packages/core/src/utils/coordinates.js"; + +export const REVIEW_VERSION = 1; +export const LAYERS = ["data", "provider", "normalization", "presentation", "runtime"] as const; +export const STAGES = { + data: ["source"], + provider: ["raw-upstream"], + normalization: ["adapted-api", "client-ranking"], + presentation: ["final-ui"], + runtime: ["engine-replay", "installed-device"], +} as const; +type Stage = (typeof STAGES)[keyof typeof STAGES][number]; +type Layer = (typeof LAYERS)[number]; +interface Entity { + label: string; + aliases?: string[]; + coordinates: [number, number]; + radiusMeters: number; + disposition: "required" | "excluded"; + address: string; + evidence: string[]; + judgment: string; +} +export interface ReviewedCase { + id: string; + query: string; + center: [number, number]; + zoom: number; + entities: Entity[]; + assessor: string; + assessedAt: string; + budgets: { + maxRank: number; + minimumRecall: number; + maxDuplicates: number; + maxWrongBranch: number; + maxForbiddenHits: number; + minimumUsefulLabels: number; + maxOverlaps: number; + maxRequests: number; + coldLatencyMs: number; + warmLatencyMs: number; + }; +} +// Pilot acceptance rules, frozen before comparison; not universal product SLAs. +const budgets = { + maxRank: 3, + minimumRecall: 1, + maxDuplicates: 0, + maxWrongBranch: 0, + maxForbiddenHits: 0, + minimumUsefulLabels: 1, + maxOverlaps: 0, + maxRequests: 1, + coldLatencyMs: 5000, + warmLatencyMs: 2000, +}; +const reviewed = { + assessor: "Codex assisted primary-source review", + assessedAt: "2026-10-07", + zoom: 15, + budgets, +}; +const rewe: Entity = { + label: "REWE", + address: "Invalidenstraße 158, Berlin", + coordinates: [13.3970838, 52.5319807], + radiusMeters: 120, + disposition: "required", + evidence: [ + "https://www.rewe.de/marktseite/berlin-mitte/1350030/rewe-markt-ackerstr-23-26-invalidenstr-158/", + "https://www.openstreetmap.org/node/348000444/history/29", + ], + judgment: + "Official address and OSM node v29 independently identify this branch; another REWE cannot substitute.", +}; +const edeka: Entity = { + label: "EDEKA", + address: "Grunerstraße 20, Berlin (ALEXA)", + coordinates: [13.416, 52.5194], + radiusMeters: 200, + disposition: "required", + evidence: ["https://www.edeka.de/maerkte/408935/"], + judgment: + "Official EDEKA Moch address verifies the tenant. Approximate mall footprint with 200m tolerance; no entrance/floor accuracy claim. A street-address result is not this business.", +}; +const media: Entity = { + label: "MediaMarkt", + address: "Grunerstraße 20, Berlin (ALEXA)", + coordinates: [13.4147909, 52.5198615], + radiusMeters: 120, + disposition: "required", + evidence: [ + "https://www.mediamarkt.de/de/store/berlin-mitte-190", + "https://www.openstreetmap.org/node/322490364/history/24", + ], + judgment: + "Official tenant address plus OSM node v24. Distinct from EDEKA at the same street address; no floor inference.", +}; +export const REVIEWED_CASES: ReviewedCase[] = [ + { + ...reviewed, + id: "business/rewe-invalidenstrasse", + query: "REWE Invalidenstraße 158", + center: [13.3970838, 52.5319807], + entities: [rewe], + }, + { + ...reviewed, + id: "business/mediamarkt-alexa", + query: "MediaMarkt Alexa", + center: [13.4147909, 52.5198615], + entities: [media], + }, + { + ...reviewed, + id: "business/edeka-alexa", + query: "EDEKA Moch Grunerstraße 20", + center: [13.416, 52.5194], + entities: [edeka], + }, + { + ...reviewed, + id: "business/alexa-tenants", + query: "", + center: [13.416, 52.5194], + entities: [media, edeka], + }, + { + ...reviewed, + id: "business/sealife-closed", + query: "SEA LIFE Berlin", + center: [13.4028, 52.5203], + entities: [ + { + label: "SEA LIFE", + address: "Berlin-Mitte historical attraction site (approximate)", + coordinates: [13.4028, 52.5203], + radiusMeters: 300, + disposition: "excluded", + evidence: ["https://lnk.bio/sealifeberlin"], + judgment: + "The attraction's own profile states permanent closure since 2024-12-13. Do not present it as an operating business; historical listings require explicit closure context. Approximate former-site coordinate.", + }, + ], + }, + ...( + [ + [ + "berlin", + "Berliner Dom", + [13.404914, 52.520407], + [13.400966, 52.519082], + "https://www.berlinerdom.de/anfahrt/", + "https://www.openstreetmap.org/way/313670734/history/78", + ], + [ + "aachen", + "Aachen Cathedral", + [6.0839, 50.7754], + [6.083957, 50.774744], + "https://www.aachenerdom.de/en/", + "https://www.openstreetmap.org/way/20470246/history/83", + ], + [ + "neuss", + "Quirinus", + [6.6916, 51.1982], + [6.693343, 51.199047], + "https://www.neuss.de/erleben/geschichte/neuss-historisch/quirinus-muenster", + "https://www.openstreetmap.org/way/28562993/history/35", + ], + [ + "monschau", + "Burg Monschau", + [6.2407, 50.5545], + [6.2397285, 50.5532282], + "https://www.monschau.de/kalender/terminanfragen/2026-08-23-burgsommer-2026-klassik-unter-sternen/", + "https://www.openstreetmap.org/node/5004288699/history/7", + ], + ] as const + ).map(([id, label, center, coordinates, url, osm]) => ({ + ...reviewed, + id: `browsing/${id}-z15`, + query: "", + center: [...center] as [number, number], + entities: [ + { + label, + aliases: + id === "berlin" + ? ["Berlin Cathedral"] + : id === "aachen" + ? ["Aachener Dom"] + : id === "monschau" + ? ["Monschau Castle"] + : [], + address: `${id} fixed baseline viewport`, + coordinates: [...coordinates] as [number, number], + radiusMeters: 200, + disposition: "required" as const, + evidence: [url, osm], + judgment: + "Official site verifies the named landmark independently of map ordering. Independent OSM geometry centroid/node with 200m tolerance, not an entrance survey; camera coordinates are not search expectations. Count fully readable useful labels and overlapping labels independently of queryRenderedFeatures counts.", + }, + ], + })), +]; + +export interface Observation { + caseId: string; + layer: Layer; + stage: Stage; + kind: "live" | "recorded" | "synthetic"; + cache: "cold" | "warm" | "uncontrolled"; + results: Array<{ label: string; coordinates: [number, number]; closed: boolean }>; + measurements: { + requests: number | null; + latencyMs: number | null; + usefulLabels: number | null; + overlaps: number | null; + }; +} +type Revision = { value: string | null; reason: string | null }; +export interface Manifest { + version: 1; + assessedAt: string; + assessor: string; + context: { + queryOrder: string[]; + cacheIsolation: "isolated" | "shared" | "unknown"; + region: string; + extractDate: Revision; + style: Revision; + deployment: Revision; + sources: { osm: Revision; overture: Revision }; + provider: { id: string; capabilities: string[] }; + configuration: { + language: string; + theme: "light" | "dark"; + viewport: [number, number]; + dpr: number; + }; + }; + observations: Observation[]; +} +function object(value: unknown): value is Record { + return !!value && typeof value === "object" && !Array.isArray(value); +} +function keys(value: Record, expected: string[]) { + return Object.keys(value).length === expected.length && expected.every((key) => key in value); +} +function text(value: unknown): value is string { + return typeof value === "string" && value.length > 0 && value.length < 500; +} +function pair(value: unknown, coordinates = true): value is [number, number] { + return ( + Array.isArray(value) && + value.length === 2 && + value.every((n) => typeof n === "number" && Number.isFinite(n)) && + (coordinates + ? Math.abs(value[0]) <= 180 && Math.abs(value[1]) <= 90 + : value.every((n) => n > 0)) + ); +} +function revision(value: unknown) { + return ( + object(value) && + keys(value, ["value", "reason"]) && + (value.value === null ? text(value.reason) : text(value.value) && value.reason === null) + ); +} +/** Only known non-secret fields survive; never echo rejected input in errors. */ +export function readManifest(value: unknown): Manifest { + const fail = () => { + throw new Error("Invalid discovery evidence manifest"); + }; + if ( + !object(value) || + !keys(value, ["version", "assessedAt", "assessor", "context", "observations"]) || + value.version !== 1 || + !text(value.assessedAt) || + !Number.isFinite(Date.parse(value.assessedAt)) || + !text(value.assessor) + ) + return fail(); + const c = value.context; + if ( + !object(c) || + !keys(c, [ + "queryOrder", + "cacheIsolation", + "region", + "extractDate", + "style", + "deployment", + "sources", + "provider", + "configuration", + ]) || + !Array.isArray(c.queryOrder) || + !c.queryOrder.every((id) => REVIEWED_CASES.some((entry) => entry.id === id)) || + typeof c.cacheIsolation !== "string" || + !["isolated", "shared", "unknown"].includes(c.cacheIsolation) || + !text(c.region) || + !revision(c.extractDate) || + !revision(c.style) || + !revision(c.deployment) || + !object(c.sources) || + !keys(c.sources, ["osm", "overture"]) || + !revision(c.sources.osm) || + !revision(c.sources.overture) || + !object(c.provider) || + !keys(c.provider, ["id", "capabilities"]) || + !text(c.provider.id) || + !Array.isArray(c.provider.capabilities) || + !c.provider.capabilities.every(text) + ) + return fail(); + const config = c.configuration; + if ( + !object(config) || + !keys(config, ["language", "theme", "viewport", "dpr"]) || + !text(config.language) || + typeof config.theme !== "string" || + !["light", "dark"].includes(config.theme) || + !pair(config.viewport, false) || + typeof config.dpr !== "number" || + !Number.isFinite(config.dpr) || + config.dpr <= 0 + ) + return fail(); + if (!Array.isArray(value.observations)) return fail(); + const identities = new Set(); + for (const o of value.observations) { + if ( + !object(o) || + !keys(o, ["caseId", "layer", "stage", "kind", "cache", "results", "measurements"]) || + !REVIEWED_CASES.some((entry) => entry.id === o.caseId) || + !LAYERS.includes(o.layer as Layer) || + typeof o.stage !== "string" || + !(STAGES[o.layer as Layer] as readonly string[]).includes(o.stage) || + typeof o.kind !== "string" || + !["live", "recorded", "synthetic"].includes(o.kind) || + typeof o.cache !== "string" || + !["cold", "warm", "uncontrolled"].includes(o.cache) || + (o.cache !== "uncontrolled" && c.cacheIsolation !== "isolated") || + !Array.isArray(o.results) || + !o.results.every( + (r) => + object(r) && + keys(r, ["label", "coordinates", "closed"]) && + text(r.label) && + pair(r.coordinates) && + typeof r.closed === "boolean", + ) || + !object(o.measurements) || + !keys(o.measurements, ["requests", "latencyMs", "usefulLabels", "overlaps"]) || + !Object.values(o.measurements).every( + (n) => n === null || (typeof n === "number" && Number.isFinite(n) && n >= 0), + ) + ) + return fail(); + if ( + [o.measurements.requests, o.measurements.usefulLabels, o.measurements.overlaps].some( + (n) => n !== null && !Number.isInteger(n), + ) + ) + return fail(); + const id = `${o.caseId}/${o.layer}`; + if (identities.has(id)) return fail(); + identities.add(id); + } + return structuredClone(value) as unknown as Manifest; +} +export function assessObservation(entry: ReviewedCase, observation: Observation) { + if (entry.id !== observation.caseId) throw new Error("Mismatched discovery case"); + const browsing = entry.id.startsWith("browsing/"); + const labelMatches = (a: string, b: string) => + a.toLocaleLowerCase("en").includes(b.toLocaleLowerCase("en")); + const matches = (r: Observation["results"][number], e: Entity) => + [e.label, ...(e.aliases ?? [])].some((label) => labelMatches(r.label, label)) && + haversineDistance(r.coordinates, e.coordinates) <= e.radiusMeters; + const required = entry.entities.filter((e) => e.disposition === "required"); + const excluded = entry.entities.filter((e) => e.disposition === "excluded"); + const ranks = required.map((e) => { + const i = observation.results.findIndex((r) => !r.closed && matches(r, e)); + return i < 0 ? null : i + 1; + }); + const matched = required.filter( + (_, i) => ranks[i] !== null && (browsing || (ranks[i] ?? Infinity) <= entry.budgets.maxRank), + ).length; + const duplicates = required.reduce( + (n, e) => n + Math.max(0, observation.results.filter((r) => matches(r, e)).length - 1), + 0, + ); + const wrongBranch = observation.results + .slice(0, browsing ? 0 : 1) + .filter( + (r) => + required.some((e) => labelMatches(r.label, e.label)) && + !required.some((e) => matches(r, e)), + ).length; + const forbiddenHits = observation.results.filter( + (r) => !r.closed && excluded.some((e) => matches(r, e)), + ).length; + const metrics = { + recall: required.length ? matched / required.length : null, + firstRank: browsing ? null : (ranks[0] ?? null), + duplicates, + wrongBranch, + forbiddenHits, + ...observation.measurements, + }; + const b = entry.budgets; + const failed = + (metrics.recall !== null && metrics.recall < b.minimumRecall) || + duplicates > b.maxDuplicates || + wrongBranch > b.maxWrongBranch || + forbiddenHits > b.maxForbiddenHits || + (metrics.requests !== null && metrics.requests > b.maxRequests) || + (observation.cache !== "uncontrolled" && + metrics.latencyMs !== null && + metrics.latencyMs > (observation.cache === "cold" ? b.coldLatencyMs : b.warmLatencyMs)) || + (observation.layer === "presentation" && + entry.id.startsWith("browsing/") && + ((metrics.usefulLabels !== null && metrics.usefulLabels < b.minimumUsefulLabels) || + (metrics.overlaps !== null && metrics.overlaps > b.maxOverlaps))); + const incomplete = + entry.id.startsWith("browsing/") && + observation.layer === "presentation" && + (metrics.usefulLabels === null || metrics.overlaps === null); + return { + caseId: entry.id, + layer: observation.layer, + stage: observation.stage, + kind: observation.kind, + cache: observation.cache, + status: failed ? "failed" : incomplete ? "unavailable" : "passed", + metrics, + }; +} +export type Assessment = ReturnType; +export function compareAssessments(before: Assessment[], after: Assessment[]) { + const changes: Array<{ + caseId: string; + layer: Layer; + metric: string; + before: number | null; + after: number | null; + }> = []; + for (const current of after) { + const previous = before.find( + (entry) => + entry.caseId === current.caseId && + entry.layer === current.layer && + entry.stage === current.stage, + ); + if (!previous) continue; + for (const metric of Object.keys(current.metrics) as Array) { + if (current.metrics[metric] !== previous.metrics[metric]) + changes.push({ + caseId: current.caseId, + layer: current.layer, + metric, + before: previous.metrics[metric], + after: current.metrics[metric], + }); + } + } + return { changes }; +} + +function reviewedCase(id: string) { + const entry = REVIEWED_CASES.find((item) => item.id === id); + if (!entry) throw new Error("Invalid discovery case"); + return entry; +} +export function buildReviewedEvidence(manifest: Manifest | null) { + const results = (manifest?.observations ?? []).map((observation) => + assessObservation(reviewedCase(observation.caseId), observation), + ); + return { + version: REVIEW_VERSION, + definitionHash: createHash("sha256").update(JSON.stringify(REVIEWED_CASES)).digest("hex"), + definitions: structuredClone(REVIEWED_CASES), + manifest, + results, + unavailable: REVIEWED_CASES.flatMap((entry) => + LAYERS.filter( + (layer) => !results.some((result) => result.caseId === entry.id && result.layer === layer), + ).map((layer) => ({ caseId: entry.id, layer })), + ), + }; +} +export type ReviewedEvidence = ReturnType; +function stable(value: unknown): string { + if (Array.isArray(value)) return JSON.stringify(value.map((item) => JSON.parse(stable(item)))); + if (object(value)) + return JSON.stringify( + Object.fromEntries( + Object.keys(value) + .sort() + .map((key) => [key, JSON.parse(stable(value[key]))]), + ), + ); + return JSON.stringify(value) ?? "null"; +} +export function validReviewedEvidence(value: unknown): value is ReviewedEvidence { + if (!object(value)) return false; + try { + const manifest = value.manifest === null ? null : readManifest(value.manifest); + return stable(value) === stable(buildReviewedEvidence(manifest)); + } catch { + return false; + } +} +export function compareReviewedEvidence(before: ReviewedEvidence, after: ReviewedEvidence) { + if (before.version !== after.version || before.definitionHash !== after.definitionHash) + throw new Error("Incompatible reviewed case definitions or budgets"); + const contextChanged = stable(before.manifest?.context) !== stable(after.manifest?.context); + const changedProviderInputs: string[] = []; + const previous = before.manifest?.observations ?? []; + const current = after.manifest?.observations ?? []; + for (const observation of [...previous, ...current]) { + if (!["data", "provider"].includes(observation.layer)) continue; + const id = `${observation.caseId}/${observation.layer}`; + const old = previous.find( + (item) => item.caseId === observation.caseId && item.layer === observation.layer, + ); + const next = current.find( + (item) => item.caseId === observation.caseId && item.layer === observation.layer, + ); + if ( + JSON.stringify(old?.results) !== JSON.stringify(next?.results) && + !changedProviderInputs.includes(id) + ) + changedProviderInputs.push(id); + } + const captureConditionsChanged = + stable( + previous + .map(({ caseId, layer, stage, kind, cache }) => ({ caseId, layer, stage, kind, cache })) + .sort((a, b) => `${a.caseId}/${a.layer}`.localeCompare(`${b.caseId}/${b.layer}`)), + ) !== + stable( + current + .map(({ caseId, layer, stage, kind, cache }) => ({ caseId, layer, stage, kind, cache })) + .sort((a, b) => `${a.caseId}/${a.layer}`.localeCompare(`${b.caseId}/${b.layer}`)), + ); + const regressions = before.results + .filter( + (result) => + result.status === "passed" && + !after.results.some( + (next) => + next.caseId === result.caseId && + next.layer === result.layer && + next.stage === result.stage && + next.status === "passed", + ), + ) + .map((result) => `${result.caseId}/${result.layer}`); + const { changes } = compareAssessments(before.results, after.results); + const lostMeasurements = changes + .filter((change) => change.before !== null && change.after === null) + .map((change) => `${change.caseId}/${change.layer}`); + return { + contextChanged, + captureConditionsChanged, + changedProviderInputs, + regressions: [...new Set([...regressions, ...lostMeasurements])], + changes, + }; +} diff --git a/scripts/discovery-eval/run.ts b/scripts/discovery-eval/run.ts index 4b61e7fbe..48668aa69 100644 --- a/scripts/discovery-eval/run.ts +++ b/scripts/discovery-eval/run.ts @@ -5,18 +5,11 @@ import { tmpdir } from "node:os"; import { dirname, join, resolve } from "node:path"; import { fileURLToPath } from "node:url"; import { CATALOG, INPUT_FILES } from "./catalog.js"; +import { parseOptions } from "./options.js"; import { compareReports, createReport, type EvalReport } from "./report.js"; +import { readManifest } from "./reviewed.js"; const root = resolve(dirname(fileURLToPath(import.meta.url)), "../.."); -function options(args: string[]) { - const result = { out: join(root, ".superpowers/eval-reports/latest"), baseline: "" }; - for (let i = 0; i < args.length; i += 2) { - if (!["--out", "--baseline"].includes(args[i]) || !args[i + 1] || args[i + 1].startsWith("--")) - throw new Error("Usage: pnpm discovery-eval [--out DIRECTORY] [--baseline REPORT.json]"); - result[args[i] === "--out" ? "out" : "baseline"] = resolve(args[i + 1]); - } - return result; -} function markdown(report: EvalReport, comparison: ReturnType | null) { const counts = report.cases.reduce>((all, entry) => { all[entry.status] = (all[entry.status] ?? 0) + 1; @@ -29,8 +22,8 @@ function markdown(report: EvalReport, comparison: ReturnType + `| ${entry.caseId} | ${entry.layer} / ${entry.stage} | ${entry.kind} / ${entry.cache} | ${entry.status} | ${JSON.stringify(entry.metrics)} |`, + ), + "", + "Absent layers (listed individually in report.json):", + "", + ...report.reviewed.unavailable.map((entry) => `- ${entry.caseId}: ${entry.layer}`), + "", ].join("\n"); } function main() { - const config = options(process.argv.slice(2)); + const config = parseOptions(process.argv.slice(2), root); + const manifest = config.evidence + ? readManifest(JSON.parse(readFileSync(config.evidence, "utf8"))) + : null; // Snapshot before output writes; reports cannot mark their own capture as dirty. const appRevision = execFileSync("git", ["rev-parse", "HEAD"], { cwd: root, @@ -96,6 +110,7 @@ function main() { CATALOG, { appRevision, workingTreeDirty, inputHashes, captureProvenance }, root, + manifest, ); const comparison = config.baseline ? compareReports(JSON.parse(readFileSync(config.baseline, "utf8")), report) @@ -111,7 +126,8 @@ function main() { test.status !== 0 || !report.runSucceeded || comparison?.regressions.length || - comparison?.runRegression + comparison?.runRegression || + comparison?.reviewed.regressions.length ) process.exitCode = 1; } finally {