diff --git a/apps/control-station/src/core/laboratory/advancedResults.ts b/apps/control-station/src/core/laboratory/advancedResults.ts index cc73fa0..9fe4efb 100644 --- a/apps/control-station/src/core/laboratory/advancedResults.ts +++ b/apps/control-station/src/core/laboratory/advancedResults.ts @@ -132,6 +132,53 @@ export interface E37AcceptanceContractResult { access: "read-only"; } +export interface E38DimensionMetric { + correct: number; + incorrect: number; + total: number; + accuracy: number; + target: number; + passed: boolean; + byStratum: Readonly>; + confusion: readonly { + reference: string; + prediction: string; + count: number; + }[]; +} + +export interface E38PerceptionBaselineResult { + resultId: string; + createdAtUtc: string | null; + sourceSessionId: string; + sourceDisplayName: string; + status: "measured-r1-source-scoped-baseline"; + profileId: string; + workerNode: string; + qualityGatePassed: boolean; + metrics: { + developmentItems: number; + validationItems: number; + terminalOutcomes: number; + accountingFraction: number; + falseFreeClaims: number; + highSeverityFailures: number; + dimensions: { + presence: E38DimensionMetric; + geometryAssociation: E38DimensionMetric; + freshness: E38DimensionMetric; + }; + }; + blockingChecks: readonly string[]; + limitations: readonly string[]; + access: "read-only"; +} + export interface AdvancedLaboratoryResults { e31: E31LaboratoryResult | null; e32: E32LaboratoryResult | null; @@ -139,6 +186,7 @@ export interface AdvancedLaboratoryResults { e34: E34TemporalLayerResult | null; e35: E35DegradationRecoveryResult | null; e37: E37AcceptanceContractResult | null; + e38: E38PerceptionBaselineResult | null; } export class AdvancedLaboratoryContractError extends Error { @@ -539,6 +587,147 @@ function parseE37(value: unknown): E37AcceptanceContractResult { }; } +function parseE38Dimension( + value: unknown, + label: string, +): E38DimensionMetric { + const source = record(value, label); + const byStratumSource = record(source.by_stratum, `${label}.by_stratum`); + if (!Array.isArray(source.confusion)) { + throw new AdvancedLaboratoryContractError( + `${label}.confusion: ожидался массив.`, + ); + } + return { + correct: integerValue(source.correct, `${label}.correct`), + incorrect: integerValue(source.incorrect, `${label}.incorrect`), + total: integerValue(source.total, `${label}.total`), + accuracy: numberValue(source.accuracy, `${label}.accuracy`), + target: numberValue(source.target, `${label}.target`), + passed: booleanValue(source.passed, `${label}.passed`), + byStratum: Object.fromEntries( + Object.entries(byStratumSource).map(([stratum, value]) => { + const metric = record(value, `${label}.by_stratum.${stratum}`); + return [ + stratum, + { + correct: integerValue( + metric.correct, + `${label}.by_stratum.${stratum}.correct`, + ), + incorrect: integerValue( + metric.incorrect, + `${label}.by_stratum.${stratum}.incorrect`, + ), + total: integerValue( + metric.total, + `${label}.by_stratum.${stratum}.total`, + ), + accuracy: numberValue( + metric.accuracy, + `${label}.by_stratum.${stratum}.accuracy`, + ), + }, + ]; + }), + ), + confusion: source.confusion.map((value, index) => { + const row = record(value, `${label}.confusion[${index}]`); + return { + reference: stringValue( + row.reference, + `${label}.confusion[${index}].reference`, + ), + prediction: stringValue( + row.prediction, + `${label}.confusion[${index}].prediction`, + ), + count: integerValue( + row.count, + `${label}.confusion[${index}].count`, + ), + }; + }), + }; +} + +function parseE38(value: unknown): E38PerceptionBaselineResult { + const item = record(value, "E38"); + const metrics = record(item.metrics, "E38.metrics"); + const dimensions = record(metrics.dimensions, "E38.metrics.dimensions"); + diagnosticAuthority(item.authority, "E38.authority"); + return { + resultId: contentId( + item.result_id, + "e38-perception-baseline", + "E38.result_id", + ), + createdAtUtc: optionalString(item.created_at_utc, "E38.created_at_utc"), + sourceSessionId: stringValue( + item.source_session_id, + "E38.source_session_id", + ), + sourceDisplayName: stringValue( + item.source_display_name, + "E38.source_display_name", + ), + status: exactString( + item.status, + "measured-r1-source-scoped-baseline", + "E38.status", + ), + profileId: stringValue(item.profile_id, "E38.profile_id"), + workerNode: stringValue(item.worker_node, "E38.worker_node"), + qualityGatePassed: booleanValue( + item.quality_gate_passed, + "E38.quality_gate_passed", + ), + metrics: { + developmentItems: integerValue( + metrics.development_items, + "E38.metrics.development_items", + ), + validationItems: integerValue( + metrics.validation_items, + "E38.metrics.validation_items", + ), + terminalOutcomes: integerValue( + metrics.terminal_outcomes, + "E38.metrics.terminal_outcomes", + ), + accountingFraction: numberValue( + metrics.accounting_fraction, + "E38.metrics.accounting_fraction", + ), + falseFreeClaims: integerValue( + metrics.false_free_claims, + "E38.metrics.false_free_claims", + ), + highSeverityFailures: integerValue( + metrics.high_severity_failures, + "E38.metrics.high_severity_failures", + ), + dimensions: { + presence: parseE38Dimension( + dimensions.presence, + "E38.metrics.dimensions.presence", + ), + geometryAssociation: parseE38Dimension( + dimensions.geometry_association, + "E38.metrics.dimensions.geometry_association", + ), + freshness: parseE38Dimension( + dimensions.freshness, + "E38.metrics.dimensions.freshness", + ), + }, + }, + blockingChecks: strings(item.blocking_checks, "E38.blocking_checks"), + limitations: strings(item.limitations, "E38.limitations"), + access: exactString(item.access, "read-only", "E38.access"), + }; +} + async function fetchOne( path: string, parser: (value: unknown) => T, @@ -563,15 +752,16 @@ export async function fetchAdvancedLaboratoryResults({ fetcher?: LaboratoryFetch; signal?: AbortSignal; } = {}): Promise { - const [e31, e32, e33, e34, e35, e37] = await Promise.all([ + const [e31, e32, e33, e34, e35, e37, e38] = await Promise.all([ fetchOne("/api/v1/laboratory/e31/results?limit=1", parseE31, fetcher, signal), fetchOne("/api/v1/laboratory/e32/results?limit=1", parseE32, fetcher, signal), fetchOne("/api/v1/laboratory/e33/results?limit=1", parseE33, fetcher, signal), fetchE34TemporalLayerResult({ fetcher, signal }), fetchE35DegradationRecoveryResult({ fetcher, signal }), fetchOne("/api/v1/laboratory/e37/results?limit=1", parseE37, fetcher, signal), + fetchOne("/api/v1/laboratory/e38/results?limit=1", parseE38, fetcher, signal), ]); - return { e31, e32, e33, e34, e35, e37 }; + return { e31, e32, e33, e34, e35, e37, e38 }; } import { fetchE34TemporalLayerResult, diff --git a/apps/control-station/src/workspaces/laboratory/AdvancedLaboratoryResult.tsx b/apps/control-station/src/workspaces/laboratory/AdvancedLaboratoryResult.tsx index 9630e4e..c46f68d 100644 --- a/apps/control-station/src/workspaces/laboratory/AdvancedLaboratoryResult.tsx +++ b/apps/control-station/src/workspaces/laboratory/AdvancedLaboratoryResult.tsx @@ -10,6 +10,7 @@ import { E33Result } from "./E33Result"; import { E34Result } from "./E34Result"; import { E35Result } from "./E35Result"; import { E37Result } from "./E37Result"; +import { E38Result } from "./E38Result"; import { RecordedReplayEvidence } from "./RecordedReplayEvidence"; export type AdvancedLaboratoryWorkId = @@ -18,7 +19,8 @@ export type AdvancedLaboratoryWorkId = | "e33-worker-shadow" | "e34-temporal-layer" | "e35-degradation-recovery" - | "e37-ravnoves-acceptance"; + | "e37-ravnoves-acceptance" + | "e38-perception-baseline"; type LaboratoryWorkspaceProps = WorkspaceRendererProps & { SpatialView: ComponentType; @@ -34,6 +36,7 @@ export function isAdvancedLaboratoryWorkId( || value === "e34-temporal-layer" || value === "e35-degradation-recovery" || value === "e37-ravnoves-acceptance" + || value === "e38-perception-baseline" ); } @@ -78,6 +81,12 @@ export function advancedLaboratoryWorkOptions( label: "LAB E37 · RAVNOVES00 acceptance R0", }); } + if (results.e38) { + options.push({ + id: "e38-perception-baseline", + label: "LAB E38 · perception quality R1", + }); + } return options; } @@ -115,6 +124,9 @@ export function AdvancedLaboratoryResult({ failedSessionId: string | null; replayError: string | null; }) { + if (workId === "e38-perception-baseline" && results.e38) { + return ; + } if (workId === "e37-ravnoves-acceptance" && results.e37) { return ; } diff --git a/apps/control-station/src/workspaces/laboratory/E38Result.tsx b/apps/control-station/src/workspaces/laboratory/E38Result.tsx new file mode 100644 index 0000000..9c3b039 --- /dev/null +++ b/apps/control-station/src/workspaces/laboratory/E38Result.tsx @@ -0,0 +1,188 @@ +import { + LaboratoryEvidence, + LaboratoryResultSummary, + LaboratorySummary, + LaboratoryWorkTemplate, +} from "../../components/laboratory/LaboratoryPresentation"; +import type { + E38DimensionMetric, + E38PerceptionBaselineResult, +} from "../../core/laboratory/advancedResults"; +import { formatNumber } from "../../presentation"; + +function percent(value: number): string { + return `${(value * 100).toLocaleString("ru-RU", { + maximumFractionDigits: 1, + })}%`; +} + +function worstStratum(metric: E38DimensionMetric): string { + const [name, value] = Object.entries(metric.byStratum) + .sort((left, right) => left[1].accuracy - right[1].accuracy)[0] ?? []; + return name && value + ? `${name}: ${percent(value.accuracy)}` + : "нет данных"; +} + +function DimensionEvidence({ + label, + metric, +}: { + label: string; + metric: E38DimensionMetric; +}) { + return ( +
+ {label} + {percent(metric.accuracy)} + + {`${formatNumber(metric.correct, 0)} / ${formatNumber(metric.total, 0)}`} + {" · слабее всего: "} + {worstStratum(metric)} + +
+ ); +} + +export function E38Result({ + rigLabel, + result, +}: { + rigLabel: string; + result: E38PerceptionBaselineResult; +}) { + const metrics = result.metrics; + const dimensions = metrics.dimensions; + const gateLabel = result.qualityGatePassed + ? "R1 gate пройден" + : "R1 требует улучшения"; + return ( + + )} + evidence={( + +
+ + + +
+ High severity + {formatNumber(metrics.highSeverityFailures, 0)} + + {"блокирующие ошибки независимо от среднего процента"} + +
+
+
+ )} + result={( + + )} + /> + ); +} diff --git a/apps/control-station/src/workspaces/laboratory/LaboratoryArchiveWorkspace.tsx b/apps/control-station/src/workspaces/laboratory/LaboratoryArchiveWorkspace.tsx index 0ca0019..57cb1be 100644 --- a/apps/control-station/src/workspaces/laboratory/LaboratoryArchiveWorkspace.tsx +++ b/apps/control-station/src/workspaces/laboratory/LaboratoryArchiveWorkspace.tsx @@ -70,6 +70,7 @@ const EMPTY_ADVANCED_RESULTS: AdvancedLaboratoryResults = { e34: null, e35: null, e37: null, + e38: null, }; function laboratoryWorkOrdinal(value: string): number { diff --git a/apps/control-station/test/advancedLaboratoryResults.test.mjs b/apps/control-station/test/advancedLaboratoryResults.test.mjs index 48018ca..94cbc46 100644 --- a/apps/control-station/test/advancedLaboratoryResults.test.mjs +++ b/apps/control-station/test/advancedLaboratoryResults.test.mjs @@ -355,6 +355,83 @@ function e37() { }; } +function e38Dimension({ + accuracy, + correct, + incorrect, + passed, +}) { + return { + correct, + incorrect, + total: 146, + accuracy, + target: 0.9, + passed, + by_stratum: { + agree: { + correct: 25, + incorrect: 4, + total: 29, + accuracy: 25 / 29, + }, + }, + confusion: [{ + reference: "object-present", + prediction: "object-present", + count: correct, + }], + }; +} + +function e38() { + return { + result_id: `e38-perception-baseline-${"8".repeat(64)}`, + created_at_utc: "2026-07-28T02:50:00Z", + source_session_id: "20260720T065719Z_viewer_live", + source_display_name: "RAVNOVES00", + status: "measured-r1-source-scoped-baseline", + profile_id: "e38-ravnoves00-r1-development-cart/v1", + worker_node: "DESKTOP-OPJ8J04", + quality_gate_passed: false, + metrics: { + development_items: 340, + validation_items: 146, + terminal_outcomes: 146, + accounting_fraction: 1, + false_free_claims: 0, + high_severity_failures: 14, + dimensions: { + presence: e38Dimension({ + accuracy: 0.821918, + correct: 120, + incorrect: 26, + passed: false, + }), + geometry_association: e38Dimension({ + accuracy: 0.815068, + correct: 119, + incorrect: 27, + passed: false, + }), + freshness: e38Dimension({ + accuracy: 0.952055, + correct: 139, + incorrect: 7, + passed: true, + }), + }, + }, + blocking_checks: [ + "presence_target_reached", + "geometry_association_target_reached", + ], + limitations: ["source-scoped"], + authority, + access: "read-only", + }; +} + before(async () => { server = await createServer({ appType: "custom", @@ -371,9 +448,9 @@ after(async () => { await server?.close(); }); -test("decodes E31–E37 from separate read-only catalogs", async () => { +test("decodes E31–E38 from separate read-only catalogs", async () => { const requests = []; - const items = [e31(), e32(), e33(), e34(), e35(), e37()]; + const items = [e31(), e32(), e33(), e34(), e35(), e37(), e38()]; const decoded = await fetchAdvancedLaboratoryResults({ fetcher: async (input, init) => { requests.push({ input: String(input), method: init?.method }); @@ -396,6 +473,9 @@ test("decodes E31–E37 from separate read-only catalogs", async () => { assert.equal(decoded.e37.metrics.reviewedItems, 486); assert.equal(decoded.e37.metrics.validationItems, 146); assert.equal(decoded.e37.qualityTargetEvaluated, false); + assert.equal(decoded.e38.metrics.dimensions.presence.accuracy, 0.821918); + assert.equal(decoded.e38.metrics.dimensions.freshness.passed, true); + assert.equal(decoded.e38.qualityGatePassed, false); assert.deepEqual(requests, [ { input: "/api/v1/laboratory/e31/results?limit=1", method: "GET" }, { input: "/api/v1/laboratory/e32/results?limit=1", method: "GET" }, @@ -403,6 +483,7 @@ test("decodes E31–E37 from separate read-only catalogs", async () => { { input: "/api/v1/laboratory/e34/results?limit=1", method: "GET" }, { input: "/api/v1/laboratory/e35/results?limit=1", method: "GET" }, { input: "/api/v1/laboratory/e37/results?limit=1", method: "GET" }, + { input: "/api/v1/laboratory/e38/results?limit=1", method: "GET" }, ]); }); @@ -418,7 +499,8 @@ test("rejects authority escalation in an accepted-looking result", async () => { : String(input).includes("/e32/") ? e32() : String(input).includes("/e33/") ? e33() : String(input).includes("/e34/") ? e34() - : String(input).includes("/e35/") ? e35() : e37(), + : String(input).includes("/e35/") ? e35() + : String(input).includes("/e37/") ? e37() : e38(), )), { status: 200 }), }), AdvancedLaboratoryContractError, diff --git a/apps/control-station/test/laboratoryProductUi.test.mjs b/apps/control-station/test/laboratoryProductUi.test.mjs index 94b2c41..872928c 100644 --- a/apps/control-station/test/laboratoryProductUi.test.mjs +++ b/apps/control-station/test/laboratoryProductUi.test.mjs @@ -30,6 +30,10 @@ const e35ResultUrl = new URL( "../src/workspaces/laboratory/E35Result.tsx", import.meta.url, ); +const e38ResultUrl = new URL( + "../src/workspaces/laboratory/E38Result.tsx", + import.meta.url, +); const e35StylesUrl = new URL( "../src/styles/e35-degradation-recovery.css", import.meta.url, @@ -217,6 +221,24 @@ test("E35 extends the canonical LAB with fault and recovery evidence", async () ); }); +test("E38 reports the frozen R1 baseline through the canonical LAB anatomy", async () => { + const [e38Source, advancedSource] = await Promise.all([ + readFile(e38ResultUrl, "utf8"), + readFile(advancedLaboratoryResultUrl, "utf8"), + ]); + + assert.match(e38Source, / { const workspacesSource = await readFile(workspacesUrl, "utf8"); diff --git a/docs/16_ARCHITECTURE_AUDIT_EXECUTION_ROADMAP.md b/docs/16_ARCHITECTURE_AUDIT_EXECUTION_ROADMAP.md index 6ef779d..3a64d78 100644 --- a/docs/16_ARCHITECTURE_AUDIT_EXECUTION_ROADMAP.md +++ b/docs/16_ARCHITECTURE_AUDIT_EXECUTION_ROADMAP.md @@ -239,6 +239,15 @@ mount identity is bound to that source session. A9 transfer therefore remains unexecuted without publishing an empty LAB. ADR 0029 retains the audit. ADR 0030 now defers TEST007/E36 and makes RAVNOVES00 source-scoped quality and product closure the active critical path. +R0 is now closed by E37 immutable result +`e37-ravnoves-acceptance-01b1efd586f747341c712d82f0907b39436a6f91ae92b1dfae987eca05fd8344`. +The first R1 baseline is E38 immutable Worker 006 result +`e38-perception-baseline-a272f82988cd9a7e071fad94c3e9fb49daf804fdcca523f853445fd3113a62b1`. +Its sealed 146-item validation measures presence at 82.2%, geometry +association at 81.5% and freshness at 95.2%, with 100% accounting, zero +false-free claims and 14 high-severity failures. R1 therefore remains open; +the next bounded iteration targets detector/background presence and +geometry-only association without changing the E37 validation set. - [x] Reproduce all 4,489 immutable E29 frames with the exact frozen profile before applying E31/E30 changes. @@ -256,7 +265,7 @@ product closure the active critical path. contract without creating an empty LAB. - [x] Defer TEST007, E36 and new collection until the RAVNOVES00 reference release candidate is accepted. -- [ ] Freeze the RAVNOVES00 task ontology, reviewed denominator and +- [x] Freeze the RAVNOVES00 task ontology, reviewed denominator and development/validation split. - [ ] Reach `>= 90%` separately for presence, geometry association and freshness decisions while retaining complete accounting and zero false-free diff --git a/docs/20_RAVNOVES00_REFERENCE_SOURCE_PRODUCT_PLAN.md b/docs/20_RAVNOVES00_REFERENCE_SOURCE_PRODUCT_PLAN.md index 17eeac8..b79d4a1 100644 --- a/docs/20_RAVNOVES00_REFERENCE_SOURCE_PRODUCT_PLAN.md +++ b/docs/20_RAVNOVES00_REFERENCE_SOURCE_PRODUCT_PLAN.md @@ -61,11 +61,28 @@ failures remain blocking even when an aggregate percentage passes. R0–R4 are the active path. R5 is intentionally deferred. +## Current evidence + +R0 is closed by immutable result +`e37-ravnoves-acceptance-01b1efd586f747341c712d82f0907b39436a6f91ae92b1dfae987eca05fd8344`. +It freezes 486 reviewed RAVNOVES00 items as 340 development and 146 sealed +validation cases, with separate presence, geometry-association and freshness +references. + +The first R1 measurement is immutable result +`e38-perception-baseline-a272f82988cd9a7e071fad94c3e9fb49daf804fdcca523f853445fd3113a62b1`. +It was trained only on the development split and executed on Worker 006. +Freshness passes at 95.2%; evidence accounting is 100% and false-free claims +remain zero. Presence is 82.2%, geometry association is 81.5% and 14 +high-severity validation cases fail, so R1 is not accepted. The next R1 +iteration is bounded to detector/background presence, geometry-only +association and the camera-only freshness tail without changing validation. + ## Experiment rules 1. RAVNOVES00 remains immutable and is never relabelled as a LAB result. 2. Every iteration creates a new LAB identity, method profile and conclusion. -3. Existing E29–E35 results remain historical evidence and are not overwritten. +3. Existing LAB results remain historical evidence and are not overwritten. 4. Metrics are computed from server-owned artifacts; the UI does not invent acceptance state. 5. The compact LAB summary explains the task, method, relevant models, diff --git a/experiments/perception/LAB_E38_REPORT_2026-07-28.md b/experiments/perception/LAB_E38_REPORT_2026-07-28.md new file mode 100644 index 0000000..8a3982c --- /dev/null +++ b/experiments/perception/LAB_E38_REPORT_2026-07-28.md @@ -0,0 +1,129 @@ +# LAB E38 — RAVNOVES00 R1 perception-quality baseline + +Date: 2026-07-28 + +Status: completed baseline; R1 quality gate not passed + +## Why this LAB exists + +E37 froze the RAVNOVES00 acceptance ontology, reviewed denominator and +development/validation split. E38 is the first planned R1 measurement against +that contract. It answers three separate questions: + +1. Does the current source-scoped evidence support that a task-relevant object + or occupied environment is present? +2. Is current geometry associated with the correct object, rejected as + non-object support, or retained as an independent occupied component? +3. Is evidence freshness represented as current, stale or unavailable? + +E38 does not evaluate navigation, planning, commands, safety or transfer to +another route or rig. + +## Immutable inputs + +| Input | Identity | +| --- | --- | +| Physical source | `20260720T065719Z_viewer_live` (`RAVNOVES00`) | +| Frozen E37 contract | `e37-ravnoves-acceptance-01b1efd586f747341c712d82f0907b39436a6f91ae92b1dfae987eca05fd8344` | +| E37 item digest | `c6e4f474f59867ce0cc86825193043245c227106400cc216a2dac0645dca54ab` | +| E30 materialization | `e30-materialization-841af926d8d28ab93538c46d8f31278a2234c4d1c12c7dc4dc296b249d59735a` | +| E30 item digest | `827b498c7b7ab520cc3998e288134ebb5a2e8ebba0532d5cc3905e1a4460b0a7` | +| E38 profile | `e38-ravnoves00-r1-development-cart/v1` | +| Profile digest | `f34012fd6fedd2a8c3e5c6866cb2e57da9cb64bd6e798772da2c338b8a9b96c1` | + +The E37 denominator contains 486 reviewed items: 340 development items and +146 sealed validation items. Validation labels are not used for fitting or +tree selection. + +## Method + +E38 joins each immutable E37 item with its immutable E30 materialization and +derives deterministic source features. Three dimension-specific shallow CART +models are trained only on the development partition: + +- presence: maximum depth 5, minimum leaf 4; +- geometry association: maximum depth 5, minimum leaf 4; +- freshness: maximum depth 3, minimum leaf 6 and source-time features enabled. + +Each validation item receives one terminal prediction for each dimension. +Targets are evaluated separately: presence `>= 90%`, geometry association +`>= 90%`, freshness `>= 90%`, accounting `100%`, false-free claims `0` and +high-severity failures `0`. + +This is an engineering-labelled source-scoped baseline, not an independent +external ground-truth benchmark. + +## Worker execution + +| Property | Value | +| --- | --- | +| Worker | `DESKTOP-OPJ8J04` (`Worker 006`) | +| Worker package | `e38-worker-package-a8b2313e07bd4d1de04db86de7728cbc21f89ae5259b2ee0adfb297eea6974c3` | +| Container | `ndc-mission-core-e38-baseline` | +| Pinned image | `nvcr.io/nvidia/tritonserver:26.06-py3@sha256:58df7489c3f2276f9591d500a012dee03e23d35543ce3c390b4c001e6bf90794` | +| Network | disabled | +| Limits | 2 CPU, 512 MiB memory, 128 PIDs | +| Filesystem | read-only runtime, explicit output mount, 64 MiB no-exec tmpfs | +| Authority | commands false; navigation/safety acceptance false | + +The result was copied back from the Worker 006 D-volume and independently +validated against its manifest and artifact digests. + +## Result + +Official result: + +`e38-perception-baseline-a272f82988cd9a7e071fad94c3e9fb49daf804fdcca523f853445fd3113a62b1` + +| Dimension | Correct | Accuracy | Target | Gate | +| --- | ---: | ---: | ---: | --- | +| Presence | 120 / 146 | 82.2% | >= 90% | fail | +| Geometry association | 119 / 146 | 81.5% | >= 90% | fail | +| Freshness | 139 / 146 | 95.2% | >= 90% | pass | +| Evidence accounting | 146 / 146 | 100% | 100% | pass | +| False-free claims | 0 | — | 0 | pass | +| High-severity failures | 14 | — | 0 | fail | + +The R1 quality gate is not passed. + +### Weakest strata + +- Presence is weakest on `geometry-only`: 25 / 38, or 65.8%. +- Geometry association is weakest on `geometry-only`: 26 / 38, or 68.4%. +- Freshness is weakest on `camera-only`: 34 / 39, or 87.2%. + +Presence errors are dominated by background/noise being promoted to +object-present and real objects being reduced to occupied environment. +Geometry errors are dominated by object-associated evidence being retained as +independent occupied geometry. + +## What E38 proves + +- RAVNOVES00 now has a reproducible first R1 baseline over a frozen holdout. +- Freshness already exceeds the source-scoped 90% target. +- All 146 validation cases receive terminal outcomes. +- No missing or rejected evidence is converted into asserted free space. +- The exact worker, immutable package, input identities, model and predictions + are retained. + +## What E38 does not prove + +- Presence and geometry association have not reached 90%. +- Fourteen high-severity validation items remain blocking. +- E38 does not prove performance on another route, camera, mount or K1 unit. +- E38 grants no navigation, command, planning or safety authority. + +## Decision and next iteration + +Keep E38 unchanged as the first measured R1 baseline. The next R1 LAB must +focus on: + +1. detector/background presence decisions; +2. separating class-bearing objects from independent `geometry-only` + occupied environment; +3. the camera-only freshness tail; +4. eliminating high-severity failures. + +The 146 validation items remain frozen. Any model or rule change is developed +only on the 340 development items and is published as a new LAB identity +before the validation gate is evaluated again. diff --git a/experiments/perception/e38_ravnoves00_r1_baseline_profile.json b/experiments/perception/e38_ravnoves00_r1_baseline_profile.json new file mode 100644 index 0000000..9ccf2b7 --- /dev/null +++ b/experiments/perception/e38_ravnoves00_r1_baseline_profile.json @@ -0,0 +1,41 @@ +{ + "authority": { + "commands_enabled": false, + "navigation_or_safety_accepted": false + }, + "model": { + "dimensions": { + "freshness": { + "include_source_time": true, + "max_depth": 3, + "min_leaf": 6 + }, + "geometry_association": { + "include_source_time": false, + "max_depth": 5, + "min_leaf": 4 + }, + "presence": { + "include_source_time": false, + "max_depth": 5, + "min_leaf": 4 + } + }, + "type": "deterministic-shallow-cart" + }, + "profile_id": "e38-ravnoves00-r1-development-cart/v1", + "schema_version": "missioncore.e38-perception-baseline-profile/v1", + "source": { + "acceptance_result_id": "e37-ravnoves-acceptance-01b1efd586f747341c712d82f0907b39436a6f91ae92b1dfae987eca05fd8344", + "display_name": "RAVNOVES00", + "materialization_id": "e30-materialization-841af926d8d28ab93538c46d8f31278a2234c4d1c12c7dc4dc296b249d59735a", + "session_id": "20260720T065719Z_viewer_live" + }, + "targets": { + "accounting_target": 1.0, + "freshness_target": 0.9, + "geometry_association_target": 0.9, + "maximum_false_free_claims": 0, + "presence_target": 0.9 + } +} diff --git a/experiments/perception/prepare_e38_worker_package.py b/experiments/perception/prepare_e38_worker_package.py new file mode 100644 index 0000000..ac918b8 --- /dev/null +++ b/experiments/perception/prepare_e38_worker_package.py @@ -0,0 +1,292 @@ +#!/usr/bin/env python3 +"""Build a minimal immutable E38 package for Worker 006.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import shutil +import uuid +from datetime import UTC, datetime +from pathlib import Path +from typing import Any + +from k1link.compute.e38_perception_baseline import ( + E38_PACKAGE_SCHEMA, + E38_PROFILE_SCHEMA, +) + +_RUNTIME_FILES = { + "runtime/k1link/__init__.py": "src/k1link/__init__.py", + "runtime/k1link/compute/__init__.py": None, + "runtime/k1link/compute/e37_acceptance_contract.py": ( + "src/k1link/compute/e37_acceptance_contract.py" + ), + "runtime/k1link/compute/e38_perception_baseline.py": ( + "src/k1link/compute/e38_perception_baseline.py" + ), + "runtime/run_e38_perception_baseline.py": ( + "experiments/perception/worker/run_e38_perception_baseline.py" + ), + "runtime/Invoke-E38PerceptionBaseline.ps1": ( + "experiments/perception/worker/Invoke-E38PerceptionBaseline.ps1" + ), +} +_GENERATED_COMPUTE_INIT = ( + '"""Minimal E38 worker projection; import contract modules explicitly."""\n' +) +_INPUT_FILES = { + "acceptance": ( + "manifest.json", + "acceptance-items.jsonl", + "acceptance-contract.json", + "run-report.json", + ), + "materialization": ("manifest.json", "materialized-items.jsonl"), +} + + +class E38WorkerPackageError(RuntimeError): + """The E38 package source or immutable package is invalid.""" + + +def build_e38_worker_package( + *, + repository_root: Path, + acceptance_root: Path, + materialization_root: Path, + profile_path: Path, + output_root: Path, +) -> Path: + """Build or verify one content-addressed E38 worker package.""" + + repository = repository_root.resolve(strict=True) + profile_source = profile_path.resolve(strict=True) + profile = _read_json(profile_source) + if profile.get("schema_version") != E38_PROFILE_SCHEMA: + raise E38WorkerPackageError("E38 package profile is incompatible") + roots = { + "acceptance": acceptance_root.resolve(strict=True), + "materialization": materialization_root.resolve(strict=True), + } + expected_ids = { + "acceptance": profile["source"]["acceptance_result_id"], + "materialization": profile["source"]["materialization_id"], + } + sources: dict[str, Path | None] = {} + for target, relative in _RUNTIME_FILES.items(): + source = None if relative is None else repository / relative + if source is not None and (not source.is_file() or source.is_symlink()): + raise E38WorkerPackageError(f"E38 runtime source is invalid: {relative}") + sources[target] = source + sources["profile.json"] = profile_source + for kind, filenames in _INPUT_FILES.items(): + root = roots[kind] + if root.name != expected_ids[kind]: + raise E38WorkerPackageError(f"E38 {kind} identity changed") + for filename in filenames: + source = root / filename + if not source.is_file() or source.is_symlink(): + raise E38WorkerPackageError(f"E38 {kind} artifact is invalid") + sources[f"input/{kind}/{root.name}/{filename}"] = source + + descriptors = [] + for relative, source in sorted(sources.items()): + payload = ( + _GENERATED_COMPUTE_INIT.encode() + if source is None + else source.read_bytes() + ) + descriptors.append( + { + "path": relative, + "byte_length": len(payload), + "sha256": hashlib.sha256(payload).hexdigest(), + } + ) + identity = { + "schema_version": E38_PACKAGE_SCHEMA, + "classification": "immutable-ravnoves00-r1-worker-input", + "source_ids": expected_ids, + "profile_sha256": _sha256(profile_source), + "artifact_paths": [row["path"] for row in descriptors], + "source_artifacts": descriptors, + "authority": { + "commands_enabled": False, + "navigation_or_safety_accepted": False, + }, + } + identity_sha256 = hashlib.sha256(_canonical_json(identity)).hexdigest() + package_id = f"e38-worker-package-{identity_sha256}" + output = output_root.expanduser().absolute() + output.mkdir(mode=0o700, parents=True, exist_ok=True) + destination = output / package_id + if destination.exists(): + validate_e38_worker_package(destination) + return destination + + staging = output / f".{package_id}.{uuid.uuid4().hex}.tmp" + staging.mkdir(mode=0o700, exist_ok=False) + try: + for relative, source in sources.items(): + target = staging / relative + target.parent.mkdir(mode=0o700, parents=True, exist_ok=True) + if source is None: + target.write_text(_GENERATED_COMPUTE_INIT, encoding="utf-8") + else: + shutil.copyfile(source, target) + artifacts = [ + { + "kind": relative, + "path": relative, + "byte_length": (staging / relative).stat().st_size, + "sha256": _sha256(staging / relative), + } + for relative in sorted(sources) + ] + manifest = { + "schema_version": E38_PACKAGE_SCHEMA, + "package_id": package_id, + "identity_sha256": identity_sha256, + "identity": identity, + "created_at_utc": datetime.now(UTC) + .isoformat(timespec="milliseconds") + .replace("+00:00", "Z"), + "artifacts": artifacts, + } + _write_json(staging / "manifest.json", manifest) + validate_e38_worker_package(staging, allow_staging=True) + os.replace(staging, destination) + except BaseException: + shutil.rmtree(staging, ignore_errors=True) + raise + validate_e38_worker_package(destination) + return destination + + +def validate_e38_worker_package( + root: Path, + *, + allow_staging: bool = False, +) -> dict[str, Any]: + """Validate package identity, exact file set, and every member digest.""" + + resolved = root.resolve(strict=True) + manifest = _read_json(resolved / "manifest.json") + identity = manifest.get("identity") + identity_sha256 = manifest.get("identity_sha256") + package_id = manifest.get("package_id") + artifacts = manifest.get("artifacts") + expected_name = ( + isinstance(package_id, str) + and ( + resolved.name == package_id + or ( + allow_staging + and resolved.name.startswith(f".{package_id}.") + and resolved.name.endswith(".tmp") + ) + ) + ) + if ( + manifest.get("schema_version") != E38_PACKAGE_SCHEMA + or not isinstance(identity, dict) + or not isinstance(identity_sha256, str) + or hashlib.sha256(_canonical_json(identity)).hexdigest() != identity_sha256 + or package_id != f"e38-worker-package-{identity_sha256}" + or not expected_name + or not isinstance(artifacts, list) + ): + raise E38WorkerPackageError("E38 worker package identity is invalid") + expected_paths = set(identity.get("artifact_paths", [])) + actual_paths = { + path.relative_to(resolved).as_posix() + for path in resolved.rglob("*") + if path.is_file() + } + if ( + not expected_paths + or actual_paths != expected_paths | {"manifest.json"} + or len(artifacts) != len(expected_paths) + ): + raise E38WorkerPackageError("E38 worker package file set changed") + observed: set[str] = set() + for row in artifacts: + if not isinstance(row, dict): + raise E38WorkerPackageError("E38 worker package artifact is invalid") + relative = row.get("path") + path = resolved / str(relative) + if ( + not isinstance(relative, str) + or relative not in expected_paths + or relative in observed + or Path(relative).is_absolute() + or ".." in Path(relative).parts + or not path.is_file() + or path.is_symlink() + or row.get("kind") != relative + or row.get("byte_length") != path.stat().st_size + or row.get("sha256") != _sha256(path) + ): + raise E38WorkerPackageError("E38 worker package artifact changed") + observed.add(relative) + if observed != expected_paths: + raise E38WorkerPackageError("E38 worker package coverage changed") + return manifest + + +def _canonical_json(value: object) -> bytes: + return json.dumps( + value, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ).encode() + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + while chunk := stream.read(1024 * 1024): + digest.update(chunk) + return digest.hexdigest() + + +def _read_json(path: Path) -> dict[str, Any]: + value = json.loads(path.read_text(encoding="utf-8-sig")) + if not isinstance(value, dict): + raise E38WorkerPackageError(f"JSON object expected: {path.name}") + return value + + +def _write_json(path: Path, value: object) -> None: + with path.open("x", encoding="utf-8", newline="\n") as stream: + json.dump(value, stream, indent=2, sort_keys=True) + stream.write("\n") + stream.flush() + os.fsync(stream.fileno()) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--repository-root", type=Path, required=True) + parser.add_argument("--acceptance", type=Path, required=True) + parser.add_argument("--materialization", type=Path, required=True) + parser.add_argument("--profile", type=Path, required=True) + parser.add_argument("--output-root", type=Path, required=True) + args = parser.parse_args() + package = build_e38_worker_package( + repository_root=args.repository_root, + acceptance_root=args.acceptance, + materialization_root=args.materialization, + profile_path=args.profile, + output_root=args.output_root, + ) + print(package) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/perception/run_e38_perception_baseline.py b/experiments/perception/run_e38_perception_baseline.py new file mode 100644 index 0000000..0cdd9fe --- /dev/null +++ b/experiments/perception/run_e38_perception_baseline.py @@ -0,0 +1,50 @@ +#!/usr/bin/env python3 +"""Build the first RAVNOVES00 R1 perception-quality baseline.""" + +from __future__ import annotations + +import argparse +import json +import os +from pathlib import Path + +from k1link.compute.e38_perception_baseline import ( + build_e38_perception_baseline, +) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--acceptance", type=Path, required=True) + parser.add_argument("--materialization", type=Path, required=True) + parser.add_argument("--profile", type=Path, required=True) + parser.add_argument("--output-root", type=Path, required=True) + parser.add_argument("--worker-node", default=os.environ.get("COMPUTERNAME")) + args = parser.parse_args() + result = build_e38_perception_baseline( + acceptance_root=args.acceptance, + materialization_root=args.materialization, + profile_path=args.profile, + output_root=args.output_root, + worker_node=args.worker_node, + ) + print( + json.dumps( + { + "result_id": result.result_id, + "result_root": str(result.result_root), + "quality_gate_passed": result.quality_gate_passed, + "metrics": result.report["metrics"], + "blocking_checks": result.report["quality_gate"][ + "blocking_checks" + ], + }, + ensure_ascii=False, + sort_keys=True, + ) + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/perception/worker/Invoke-E38PerceptionBaseline.ps1 b/experiments/perception/worker/Invoke-E38PerceptionBaseline.ps1 new file mode 100644 index 0000000..64f7630 --- /dev/null +++ b/experiments/perception/worker/Invoke-E38PerceptionBaseline.ps1 @@ -0,0 +1,138 @@ +[CmdletBinding()] +param( + [Parameter(Mandatory = $true)] + [string]$PackageRoot, + [string]$OutputRoot = "D:\NDC_MISSIONCORE\runtime\derived\e38-baseline", + [string]$ContainerImage = "nvcr.io/nvidia/tritonserver:26.06-py3@sha256:58df7489c3f2276f9591d500a012dee03e23d35543ce3c390b4c001e6bf90794", + [ValidateRange(1, 1000)] + [int]$FreeGiBFloor = 300 +) + +$ErrorActionPreference = "Stop" +$ProgressPreference = "SilentlyContinue" + +function Assert-LastExitCode([string]$Operation) { + if ($LASTEXITCODE -ne 0) { + throw "$Operation failed with exit code $LASTEXITCODE" + } +} + +function Resolve-DDirectory([string]$Path, [string]$Label) { + $item = Get-Item -LiteralPath (Resolve-Path -LiteralPath $Path).Path -Force + $root = [IO.Path]::GetPathRoot($item.FullName).TrimEnd("\") + if ( + -not $item.PSIsContainer -or + ($item.Attributes -band [IO.FileAttributes]::ReparsePoint) -or + $root -ine "D:" + ) { + throw "$Label must be a real D: directory" + } + return $item.FullName +} + +function Convert-ToDockerPath([string]$Path) { + return $Path.Replace("\", "/") +} + +function Assert-FreeSpace([string]$Phase) { + $free = [int64](Get-PSDrive -Name D).Free + $floor = [int64]$FreeGiBFloor * 1GB + Write-Host ( + "DISK_GUARD PHASE={0} DRIVE=D FREE_BYTES={1} FREE_GIB={2} FLOOR_GIB={3}" -f + $Phase, $free, [math]::Round($free / 1GB, 3), $FreeGiBFloor + ) + if ($free -lt ($floor + 1GB)) { + throw "D: lacks the guarded E38 reserve during $Phase" + } + return $free +} + +$package = Resolve-DDirectory $PackageRoot "E38 package" +$packageManifestPath = Join-Path $package "manifest.json" +if (-not (Test-Path -LiteralPath $packageManifestPath -PathType Leaf)) { + throw "E38 package manifest is missing" +} +$packageManifest = Get-Content -LiteralPath $packageManifestPath -Raw | + ConvertFrom-Json +if ( + $packageManifest.schema_version -ne "missioncore.e38-worker-package/v1" -or + $packageManifest.package_id -ne (Split-Path $package -Leaf) -or + $packageManifest.package_id -notmatch "^e38-worker-package-[a-f0-9]{64}$" +) { + throw "E38 package manifest is incompatible" +} + +if (-not (Test-Path -LiteralPath $OutputRoot)) { + $null = New-Item -ItemType Directory -Path $OutputRoot +} +$output = Resolve-DDirectory $OutputRoot "E38 output root" +$freeBefore = Assert-FreeSpace "preflight" + +& docker image inspect $ContainerImage *> $null +Assert-LastExitCode "Pinned E38 container image inspection" + +$dockerPackage = Convert-ToDockerPath $package +$dockerOutput = Convert-ToDockerPath $output +$packageName = Split-Path $package -Leaf +$containerPackage = "/opt/e38-input/$packageName" +$command = @( + "run", "--rm", + "--name", "ndc-mission-core-e38-baseline", + "--network", "none", + "--read-only", + "--security-opt", "no-new-privileges:true", + "--cap-drop", "ALL", + "--pids-limit", "128", + "--memory", "512m", + "--memory-swap", "512m", + "--cpus", "2", + "--tmpfs", "/tmp:rw,noexec,nosuid,size=64m", + "-e", "PYTHONDONTWRITEBYTECODE=1", + "-e", ("PYTHONPATH={0}/runtime" -f $containerPackage), + "-e", ("E38_WORKER_NODE={0}" -f $env:COMPUTERNAME), + "-v", ("{0}:{1}:ro" -f $dockerPackage, $containerPackage), + "-v", ("{0}:/output:rw" -f $dockerOutput), + "--entrypoint", "python3", + $ContainerImage, + ("{0}/runtime/run_e38_perception_baseline.py" -f $containerPackage), + "--package", $containerPackage, + "--output-root", "/output" +) + +Write-Output ("PACKAGE_ID={0}" -f $packageManifest.package_id) +Write-Output ("PACKAGE_IDENTITY_SHA256={0}" -f $packageManifest.identity_sha256) +Write-Output ("CONTAINER_IMAGE={0}" -f $ContainerImage) +& docker @command +Assert-LastExitCode "E38 perception baseline" + +$matches = @( + Get-ChildItem -LiteralPath $output -Directory -Filter "e38-perception-baseline-*" | + Where-Object { + $manifestPath = Join-Path $_.FullName "manifest.json" + if (-not (Test-Path -LiteralPath $manifestPath -PathType Leaf)) { + return $false + } + $manifest = Get-Content -LiteralPath $manifestPath -Raw | + ConvertFrom-Json + return ( + $manifest.schema_version -eq + "missioncore.e38-perception-baseline/v1" -and + $manifest.acceptance_state -eq + "completed-r1-source-scoped-baseline" -and + $manifest.identity.execution.worker_node -eq $env:COMPUTERNAME + ) + } +) +if ($matches.Count -ne 1) { + throw "E38 immutable result could not be resolved uniquely" +} +$resultRoot = $matches[0].FullName +$resultManifest = Get-Content -LiteralPath ( + Join-Path $resultRoot "manifest.json" +) -Raw | ConvertFrom-Json +$freeAfter = Assert-FreeSpace "completed" +Write-Output ("RESULT_ROOT={0}" -f $resultRoot) +Write-Output ("RESULT_ID={0}" -f $resultManifest.result_id) +Write-Output ("QUALITY_GATE_PASSED={0}" -f $resultManifest.quality_gate_passed) +Write-Output ("DISK_FREE_BYTES_BEFORE={0}" -f $freeBefore) +Write-Output ("DISK_FREE_BYTES_AFTER={0}" -f $freeAfter) diff --git a/experiments/perception/worker/run_e38_perception_baseline.py b/experiments/perception/worker/run_e38_perception_baseline.py new file mode 100644 index 0000000..4f99330 --- /dev/null +++ b/experiments/perception/worker/run_e38_perception_baseline.py @@ -0,0 +1,48 @@ +#!/usr/bin/env python3 +"""Execute a packaged E38 baseline inside the pinned Worker 006 container.""" + +from __future__ import annotations + +import argparse +import json +import os +from pathlib import Path + +from k1link.compute.e38_perception_baseline import ( + build_e38_perception_baseline, +) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--package", type=Path, required=True) + parser.add_argument("--output-root", type=Path, required=True) + args = parser.parse_args() + package = args.package.resolve(strict=True) + manifest = json.loads((package / "manifest.json").read_text(encoding="utf-8")) + source = manifest["identity"]["source_ids"] + result = build_e38_perception_baseline( + acceptance_root=package / "input" / "acceptance" / source["acceptance"], + materialization_root=( + package / "input" / "materialization" / source["materialization"] + ), + profile_path=package / "profile.json", + output_root=args.output_root, + worker_node=os.environ.get("E38_WORKER_NODE"), + ) + print( + json.dumps( + { + "result_id": result.result_id, + "quality_gate_passed": result.quality_gate_passed, + "metrics": result.report["metrics"], + }, + ensure_ascii=False, + sort_keys=True, + ) + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/k1link/compute/e38_perception_baseline.py b/src/k1link/compute/e38_perception_baseline.py new file mode 100644 index 0000000..1d6f79a --- /dev/null +++ b/src/k1link/compute/e38_perception_baseline.py @@ -0,0 +1,830 @@ +"""Development-trained RAVNOVES00 R1 perception-quality baseline. + +E38 consumes the frozen E37 contract without changing its denominator or +validation split. A small deterministic decision tree is fitted only on the +development partition and is then evaluated once on the sealed validation +partition. The result remains source-scoped and diagnostic. +""" + +from __future__ import annotations + +import hashlib +import json +import math +import os +import shutil +import uuid +from collections import Counter, defaultdict +from dataclasses import dataclass +from datetime import UTC, datetime +from pathlib import Path +from typing import Any, Final + +from k1link.compute.e37_acceptance_contract import ( + E37_ITEMS_NAME, + E37AcceptanceContractError, + read_e37_acceptance_contract, +) + +E38_PROFILE_SCHEMA: Final = "missioncore.e38-perception-baseline-profile/v1" +E38_PACKAGE_SCHEMA: Final = "missioncore.e38-worker-package/v1" +E38_RESULT_SCHEMA: Final = "missioncore.e38-perception-baseline/v1" +E38_PREDICTION_SCHEMA: Final = "missioncore.e38-perception-prediction/v1" +E38_MODEL_SCHEMA: Final = "missioncore.e38-development-cart-model/v1" +E38_REPORT_SCHEMA: Final = "missioncore.e38-perception-baseline-report/v1" + +E38_PREDICTIONS_NAME: Final = "predictions.jsonl" +E38_MODEL_NAME: Final = "development-model.json" +E38_REPORT_NAME: Final = "run-report.json" +E38_MANIFEST_NAME: Final = "manifest.json" + +_MATERIALIZATION_SCHEMA: Final = "missioncore.e30-evidence-materialization/v2" +_SOURCE_SESSION_ID: Final = "20260720T065719Z_viewer_live" +_SOURCE_DISPLAY_NAME: Final = "RAVNOVES00" +_DIMENSIONS: Final = ("presence", "geometry_association", "freshness") +_CATEGORICAL_FEATURES: Final = { + "stratum": ("agree", "camera-only", "conflict", "geometry-only", "unknown"), + "range": ("near", "middle", "far", "unavailable"), + "geometry": ( + "agree", + "conflict", + "single-source-camera", + "single-source-geometry", + "unknown", + "unavailable", + ), +} +_AUTHORITY: Final = { + "commands_enabled": False, + "navigation_or_safety_accepted": False, +} + + +class E38PerceptionBaselineError(RuntimeError): + """The E38 profile, immutable inputs, or result is invalid.""" + + +@dataclass(frozen=True, slots=True) +class E38PerceptionBaseline: + result_id: str + result_root: Path + manifest: dict[str, Any] + report: dict[str, Any] + model: dict[str, Any] + + @property + def quality_gate_passed(self) -> bool: + return self.report.get("quality_gate", {}).get("passed") is True + + +def build_e38_perception_baseline( + *, + acceptance_root: Path, + materialization_root: Path, + profile_path: Path, + output_root: Path, + worker_node: str | None = None, +) -> E38PerceptionBaseline: + """Fit on E37 development rows and evaluate the sealed validation rows.""" + + profile_file = profile_path.resolve(strict=True) + profile = _read_json(profile_file) + _validate_profile(profile) + acceptance_path = acceptance_root.resolve(strict=True) + materialization_path = materialization_root.resolve(strict=True) + try: + acceptance = read_e37_acceptance_contract(acceptance_path) + except E37AcceptanceContractError as exc: + raise E38PerceptionBaselineError("E38 E37 contract is invalid") from exc + if acceptance.result_id != profile["source"]["acceptance_result_id"]: + raise E38PerceptionBaselineError("E38 acceptance identity changed") + + acceptance_items_path = acceptance_path / E37_ITEMS_NAME + acceptance_rows = _read_jsonl(acceptance_items_path) + materialization_rows, materialization_binding = _load_materialization( + materialization_path, + acceptance.manifest, + profile, + ) + if ( + len(acceptance_rows) != 486 + or len(materialization_rows) != 486 + or {row.get("item_id") for row in acceptance_rows} + != {row.get("item_id") for row in materialization_rows} + ): + raise E38PerceptionBaselineError("E38 denominator accounting differs") + + identity = { + "schema_version": E38_RESULT_SCHEMA, + "source": { + "session_id": _SOURCE_SESSION_ID, + "display_name": _SOURCE_DISPLAY_NAME, + "acceptance_result_id": acceptance.result_id, + "acceptance_identity_sha256": acceptance.manifest["identity_sha256"], + "acceptance_items_sha256": _sha256(acceptance_items_path), + **materialization_binding, + }, + "profile": { + "profile_id": profile["profile_id"], + "sha256": _sha256(profile_file), + }, + "execution": { + "class": "deterministic-development-trained-validation-evaluation", + "worker_node": worker_node or "unbound-local", + }, + "authority": _AUTHORITY, + } + identity_sha256 = hashlib.sha256(_canonical_json(identity)).hexdigest() + result_id = f"e38-perception-baseline-{identity_sha256}" + destination = output_root.expanduser().absolute() / result_id + if destination.exists(): + return read_e38_perception_baseline(destination) + + materialization_by_id = { + str(row["item_id"]): row for row in materialization_rows + } + joined: list[dict[str, Any]] = [] + for acceptance_row in acceptance_rows: + item_id = str(acceptance_row.get("item_id")) + reference = acceptance_row.get("reference") + if ( + acceptance_row.get("split") not in {"development", "validation"} + or acceptance_row.get("severity") not in {"standard", "medium", "high"} + or not isinstance(reference, dict) + or set(reference) != set(_DIMENSIONS) + ): + raise E38PerceptionBaselineError("E38 acceptance row is invalid") + joined.append( + { + "acceptance": acceptance_row, + "features": _feature_vector( + acceptance_row, + materialization_by_id[item_id], + ), + } + ) + + model_dimensions: dict[str, Any] = {} + for dimension in _DIMENSIONS: + dimension_profile = profile["model"]["dimensions"][dimension] + feature_names = _dimension_features( + joined, + include_source_time=dimension_profile["include_source_time"], + ) + development = [ + ( + _selected_features(row["features"], feature_names), + str(row["acceptance"]["reference"][dimension]), + ) + for row in joined + if row["acceptance"]["split"] == "development" + ] + tree = _train_tree( + development, + feature_names=feature_names, + max_depth=int(dimension_profile["max_depth"]), + min_leaf=int(dimension_profile["min_leaf"]), + ) + model_dimensions[dimension] = { + "feature_names": feature_names, + "max_depth": dimension_profile["max_depth"], + "min_leaf": dimension_profile["min_leaf"], + "tree": tree, + } + + predictions: list[dict[str, Any]] = [] + for row in joined: + acceptance_row = row["acceptance"] + predicted = { + dimension: _predict_tree( + model_dimensions[dimension]["tree"], + row["features"], + ) + for dimension in _DIMENSIONS + } + predictions.append( + { + "schema_version": E38_PREDICTION_SCHEMA, + "sequence": acceptance_row["sequence"], + "item_id": acceptance_row["item_id"], + "review_key": acceptance_row["review_key"], + "source_frame_index": acceptance_row["source_frame_index"], + "source_stratum": acceptance_row["source_stratum"], + "severity": acceptance_row["severity"], + "split": acceptance_row["split"], + "prediction": predicted, + "reference": acceptance_row["reference"], + "scored": acceptance_row["split"] == "validation", + "authority": _AUTHORITY, + } + ) + + validation_predictions = [ + row for row in predictions if row["split"] == "validation" + ] + development_count = sum(row["split"] == "development" for row in predictions) + validation_count = len(validation_predictions) + if development_count != 340 or validation_count != 146: + raise E38PerceptionBaselineError("E38 frozen split changed") + dimension_metrics = { + dimension: _dimension_metrics( + validation_predictions, + dimension=dimension, + target=float(profile["targets"][f"{dimension}_target"]), + ) + for dimension in _DIMENSIONS + } + high_severity_failures = sum( + row["severity"] == "high" + and any( + row["prediction"][dimension] != row["reference"][dimension] + for dimension in _DIMENSIONS + ) + for row in validation_predictions + ) + accounting_fraction = ( + len(validation_predictions) / validation_count if validation_count else 0.0 + ) + false_free_claims = sum( + value == "free" + for row in predictions + for value in row["prediction"].values() + ) + gate_checks = { + "presence_target_reached": dimension_metrics["presence"]["passed"], + "geometry_association_target_reached": dimension_metrics[ + "geometry_association" + ]["passed"], + "freshness_target_reached": dimension_metrics["freshness"]["passed"], + "validation_accounting_complete": math.isclose(accounting_fraction, 1.0), + "false_free_claims_zero": false_free_claims == 0, + "high_severity_failures_zero": high_severity_failures == 0, + "authority_remains_diagnostic": True, + } + quality_gate_passed = all(gate_checks.values()) + model_document = { + "schema_version": E38_MODEL_SCHEMA, + "profile_id": profile["profile_id"], + "training_split": "development", + "training_items": development_count, + "validation_labels_used_for_training": False, + "dimensions": model_dimensions, + "authority": _AUTHORITY, + } + report = { + "schema_version": E38_REPORT_SCHEMA, + "result_id": result_id, + "identity_sha256": identity_sha256, + "status": "measured-r1-source-scoped-baseline", + "source_session_id": _SOURCE_SESSION_ID, + "source_display_name": _SOURCE_DISPLAY_NAME, + "profile_id": profile["profile_id"], + "execution": identity["execution"], + "metrics": { + "development_items": development_count, + "validation_items": validation_count, + "terminal_outcomes": validation_count, + "accounting_fraction": round(accounting_fraction, 6), + "false_free_claims": false_free_claims, + "high_severity_failures": high_severity_failures, + "dimensions": dimension_metrics, + }, + "quality_gate": { + "passed": quality_gate_passed, + "checks": gate_checks, + "blocking_checks": [ + name for name, passed in gate_checks.items() if not passed + ], + }, + "decision": { + "r1_baseline_measured": True, + "accepted_for_release": quality_gate_passed, + "next_gate": ( + "R2 source-scoped temporal product state" + if quality_gate_passed + else "R1 detector presence and geometry-association improvement" + ), + }, + "limitations": [ + ( + "the model is trained and evaluated only on the source-scoped " + "RAVNOVES00 engineering-reviewed contract" + ), + "validation labels are evaluation-only and never used to fit a tree", + ( + "the result is not independent ground truth and does not prove " + "another route, camera, rig or mount" + ), + "navigation, command and safety authority remain false", + ], + "authority": _AUTHORITY, + } + + destination.parent.mkdir(mode=0o700, parents=True, exist_ok=True) + staging = destination.parent / f".{result_id}.{uuid.uuid4().hex}.tmp" + staging.mkdir(mode=0o700, exist_ok=False) + try: + _write_jsonl(staging / E38_PREDICTIONS_NAME, predictions) + _write_json(staging / E38_MODEL_NAME, model_document) + _write_json(staging / E38_REPORT_NAME, report) + artifacts = [ + _artifact(staging / E38_PREDICTIONS_NAME, "sealed-evaluation"), + _artifact(staging / E38_MODEL_NAME, "development-trained-model"), + _artifact(staging / E38_REPORT_NAME, "quality-report"), + ] + manifest = { + "schema_version": E38_RESULT_SCHEMA, + "result_id": result_id, + "identity_sha256": identity_sha256, + "identity": identity, + "created_at_utc": _utc_now(), + "acceptance_state": "completed-r1-source-scoped-baseline", + "quality_gate_passed": quality_gate_passed, + "artifacts": artifacts, + } + _write_json(staging / E38_MANIFEST_NAME, manifest) + os.replace(staging, destination) + except BaseException: + shutil.rmtree(staging, ignore_errors=True) + raise + return read_e38_perception_baseline(destination) + + +def read_e38_perception_baseline(root: Path) -> E38PerceptionBaseline: + """Validate and read one immutable E38 result.""" + + resolved = root.resolve(strict=True) + manifest = _read_json(resolved / E38_MANIFEST_NAME) + identity = manifest.get("identity") + identity_sha256 = manifest.get("identity_sha256") + if ( + manifest.get("schema_version") != E38_RESULT_SCHEMA + or not isinstance(identity, dict) + or not isinstance(identity_sha256, str) + or hashlib.sha256(_canonical_json(identity)).hexdigest() != identity_sha256 + or manifest.get("result_id") != f"e38-perception-baseline-{identity_sha256}" + or resolved.name != manifest.get("result_id") + or manifest.get("acceptance_state") + != "completed-r1-source-scoped-baseline" + or identity.get("authority") != _AUTHORITY + ): + raise E38PerceptionBaselineError("E38 result identity is invalid") + expected = { + E38_PREDICTIONS_NAME: "sealed-evaluation", + E38_MODEL_NAME: "development-trained-model", + E38_REPORT_NAME: "quality-report", + } + artifacts = manifest.get("artifacts") + if not isinstance(artifacts, list) or len(artifacts) != len(expected): + raise E38PerceptionBaselineError("E38 artifact catalog is invalid") + for row in artifacts: + if not isinstance(row, dict): + raise E38PerceptionBaselineError("E38 artifact descriptor is invalid") + name = row.get("path") + path = resolved / str(name) + if ( + name not in expected + or row.get("role") != expected[name] + or not path.is_file() + or path.is_symlink() + or row.get("byte_length") != path.stat().st_size + or row.get("sha256") != _sha256(path) + ): + raise E38PerceptionBaselineError("E38 artifact content changed") + report = _read_json(resolved / E38_REPORT_NAME) + model = _read_json(resolved / E38_MODEL_NAME) + if ( + report.get("schema_version") != E38_REPORT_SCHEMA + or report.get("result_id") != resolved.name + or report.get("identity_sha256") != identity_sha256 + or report.get("status") != "measured-r1-source-scoped-baseline" + or model.get("schema_version") != E38_MODEL_SCHEMA + or model.get("validation_labels_used_for_training") is not False + or manifest.get("quality_gate_passed") + is not report.get("quality_gate", {}).get("passed") + ): + raise E38PerceptionBaselineError("E38 report or model is invalid") + return E38PerceptionBaseline( + result_id=resolved.name, + result_root=resolved, + manifest=manifest, + report=report, + model=model, + ) + + +def _load_materialization( + root: Path, + acceptance_manifest: dict[str, Any], + profile: dict[str, Any], +) -> tuple[list[dict[str, Any]], dict[str, Any]]: + manifest_path = root / "manifest.json" + index_path = root / "materialized-items.jsonl" + manifest = _read_json(manifest_path) + binding = ( + acceptance_manifest.get("identity", {}) + .get("reviewed_substrate", {}) + ) + if ( + manifest.get("schema_version") != _MATERIALIZATION_SCHEMA + or root.name != profile["source"]["materialization_id"] + or manifest.get("result_id") != root.name + or root.name != binding.get("materialization_id") + or manifest.get("identity_sha256") + != binding.get("materialization_identity_sha256") + or _sha256(manifest_path) + != binding.get("materialization_manifest_sha256") + or not index_path.is_file() + or index_path.is_symlink() + or _sha256(index_path) != binding.get("materialization_index_sha256") + ): + raise E38PerceptionBaselineError("E38 materialization identity changed") + rows = _read_jsonl(index_path) + return rows, { + "materialization_id": root.name, + "materialization_identity_sha256": manifest["identity_sha256"], + "materialization_index_sha256": _sha256(index_path), + } + + +def _feature_vector( + acceptance: dict[str, Any], + materialization: dict[str, Any], +) -> dict[str, float]: + snapshot = _object(materialization.get("e29_snapshot"), "E38 snapshot") + evidence = _object( + materialization.get("materialization"), + "E38 materialization evidence", + ) + bounds = snapshot.get("bounds_map_xyz_m") + height = snapshot.get("height_range_m") + features = { + "detector_score": _number_or(evidence.get("detector_score"), -1.0), + "selected_points": _number_or(evidence.get("selected_point_count"), 0.0), + "rejected_points": _number_or( + evidence.get("rejected_candidate_point_count"), + 0.0, + ), + "candidate_points": _number_or(evidence.get("candidate_point_count"), 0.0), + "camera_front_points": _number_or( + evidence.get("camera_front_point_count"), + 0.0, + ), + "projected_points": _number_or( + evidence.get("projected_point_count"), + 0.0, + ), + "frame_points": _number_or(evidence.get("frame_point_count"), 0.0), + "point_count": _number_or(snapshot.get("point_count"), 0.0), + "voxel_count": _number_or(snapshot.get("voxel_count"), 0.0), + "nearest_range_m": _number_or(snapshot.get("nearest_range_m"), -1.0), + "height_span_m": _span(height, 0), + "bounds_span_x_m": _span(bounds, 0), + "bounds_span_y_m": _span(bounds, 1), + "bounds_span_z_m": _span(bounds, 2), + "session_seconds": _number_or(acceptance.get("session_seconds"), -1.0), + "source_frame_index": _number_or( + acceptance.get("source_frame_index"), + -1.0, + ), + } + categorical = { + "stratum": materialization.get("stratum"), + "range": materialization.get("range_bucket"), + "geometry": snapshot.get("geometry_status"), + } + for prefix, values in _CATEGORICAL_FEATURES.items(): + observed = str(categorical[prefix]) + for value in values: + features[f"{prefix}={value}"] = float(observed == value) + return features + + +def _dimension_features( + rows: list[dict[str, Any]], + *, + include_source_time: bool, +) -> list[str]: + features = sorted( + { + name + for row in rows + for name in row["features"] + if include_source_time + or name not in {"session_seconds", "source_frame_index"} + } + ) + if not features: + raise E38PerceptionBaselineError("E38 feature set is empty") + return features + + +def _selected_features( + values: dict[str, float], + names: list[str], +) -> dict[str, float]: + return {name: values.get(name, 0.0) for name in names} + + +def _train_tree( + rows: list[tuple[dict[str, float], str]], + *, + feature_names: list[str], + max_depth: int, + min_leaf: int, + depth: int = 0, +) -> dict[str, Any]: + if not rows: + raise E38PerceptionBaselineError("E38 tree has no training rows") + labels = [label for _, label in rows] + prediction = sorted(Counter(labels).items(), key=lambda item: (-item[1], item[0]))[ + 0 + ][0] + node: dict[str, Any] = { + "prediction": prediction, + "samples": len(rows), + "distribution": dict(sorted(Counter(labels).items())), + } + if ( + depth >= max_depth + or len(set(labels)) == 1 + or len(rows) < 2 * min_leaf + ): + return node + base_impurity = _gini(labels) + best: tuple[ + float, + str, + float, + list[tuple[dict[str, float], str]], + list[tuple[dict[str, float], str]], + ] | None = None + for feature in feature_names: + values = sorted({float(features.get(feature, 0.0)) for features, _ in rows}) + for left_value, right_value in zip(values, values[1:], strict=False): + threshold = (left_value + right_value) / 2.0 + left_rows = [ + row for row in rows if float(row[0].get(feature, 0.0)) <= threshold + ] + right_rows = [ + row for row in rows if float(row[0].get(feature, 0.0)) > threshold + ] + if len(left_rows) < min_leaf or len(right_rows) < min_leaf: + continue + impurity = ( + len(left_rows) * _gini([label for _, label in left_rows]) + + len(right_rows) * _gini([label for _, label in right_rows]) + ) / len(rows) + gain = base_impurity - impurity + candidate = (gain, feature, threshold, left_rows, right_rows) + if best is None or _better_split(candidate, best): + best = candidate + if best is None or best[0] <= 1e-12: + return node + node.update( + { + "feature": best[1], + "threshold": round(best[2], 12), + "left": _train_tree( + best[3], + feature_names=feature_names, + max_depth=max_depth, + min_leaf=min_leaf, + depth=depth + 1, + ), + "right": _train_tree( + best[4], + feature_names=feature_names, + max_depth=max_depth, + min_leaf=min_leaf, + depth=depth + 1, + ), + } + ) + return node + + +def _better_split( + candidate: tuple[float, str, float, Any, Any], + current: tuple[float, str, float, Any, Any], +) -> bool: + if candidate[0] > current[0] + 1e-12: + return True + if abs(candidate[0] - current[0]) <= 1e-12: + return (candidate[1], candidate[2]) < (current[1], current[2]) + return False + + +def _predict_tree(tree: dict[str, Any], features: dict[str, float]) -> str: + node = tree + while "feature" in node: + feature = str(node["feature"]) + threshold = float(node["threshold"]) + node = ( + _object(node.get("left"), "E38 tree left") + if features.get(feature, 0.0) <= threshold + else _object(node.get("right"), "E38 tree right") + ) + prediction = node.get("prediction") + if not isinstance(prediction, str) or not prediction: + raise E38PerceptionBaselineError("E38 tree prediction is invalid") + return prediction + + +def _dimension_metrics( + rows: list[dict[str, Any]], + *, + dimension: str, + target: float, +) -> dict[str, Any]: + confusion: Counter[tuple[str, str]] = Counter() + stratum_counts: dict[str, Counter[str]] = defaultdict(Counter) + correct = 0 + for row in rows: + reference = str(row["reference"][dimension]) + prediction = str(row["prediction"][dimension]) + confusion[(reference, prediction)] += 1 + matched = reference == prediction + correct += matched + stratum_counts[str(row["source_stratum"])][ + "correct" if matched else "incorrect" + ] += 1 + total = len(rows) + accuracy = correct / total if total else 0.0 + return { + "correct": correct, + "incorrect": total - correct, + "total": total, + "accuracy": round(accuracy, 6), + "target": target, + "passed": accuracy >= target, + "confusion": [ + { + "reference": reference, + "prediction": prediction, + "count": count, + } + for (reference, prediction), count in sorted( + confusion.items(), + key=lambda item: (-item[1], item[0]), + ) + ], + "by_stratum": { + stratum: { + "correct": counts["correct"], + "incorrect": counts["incorrect"], + "total": sum(counts.values()), + "accuracy": round( + counts["correct"] / sum(counts.values()), + 6, + ), + } + for stratum, counts in sorted(stratum_counts.items()) + }, + } + + +def _gini(labels: list[str]) -> float: + counts = Counter(labels) + total = len(labels) + return 1.0 - sum((count / total) ** 2 for count in counts.values()) + + +def _span(value: object, axis: int) -> float: + if not isinstance(value, list) or len(value) != 2: + return -1.0 + if axis == 0 and all(isinstance(item, (int, float)) for item in value): + return float(value[1]) - float(value[0]) + if not all( + isinstance(item, list) + and len(item) > axis + and isinstance(item[axis], (int, float)) + for item in value + ): + return -1.0 + return float(value[1][axis]) - float(value[0][axis]) + + +def _number_or(value: object, fallback: float) -> float: + if isinstance(value, (int, float)) and not isinstance(value, bool): + parsed = float(value) + if math.isfinite(parsed): + return parsed + return fallback + + +def _validate_profile(profile: dict[str, Any]) -> None: + source = _object(profile.get("source"), "E38 source") + model = _object(profile.get("model"), "E38 model") + dimensions = _object(model.get("dimensions"), "E38 model dimensions") + targets = _object(profile.get("targets"), "E38 targets") + if ( + profile.get("schema_version") != E38_PROFILE_SCHEMA + or profile.get("profile_id") + != "e38-ravnoves00-r1-development-cart/v1" + or source.get("session_id") != _SOURCE_SESSION_ID + or source.get("display_name") != _SOURCE_DISPLAY_NAME + or not isinstance(source.get("acceptance_result_id"), str) + or not str(source["acceptance_result_id"]).startswith( + "e37-ravnoves-acceptance-" + ) + or not isinstance(source.get("materialization_id"), str) + or not str(source["materialization_id"]).startswith("e30-materialization-") + or model.get("type") != "deterministic-shallow-cart" + or set(dimensions) != set(_DIMENSIONS) + or any( + not isinstance(dimensions[name], dict) + or not isinstance(dimensions[name].get("include_source_time"), bool) + or not isinstance(dimensions[name].get("max_depth"), int) + or not 1 <= dimensions[name]["max_depth"] <= 8 + or not isinstance(dimensions[name].get("min_leaf"), int) + or not 2 <= dimensions[name]["min_leaf"] <= 32 + for name in _DIMENSIONS + ) + or any(targets.get(f"{name}_target") != 0.9 for name in _DIMENSIONS) + or targets.get("accounting_target") != 1.0 + or targets.get("maximum_false_free_claims") != 0 + or profile.get("authority") != _AUTHORITY + ): + raise E38PerceptionBaselineError("E38 profile contract changed") + + +def _artifact(path: Path, role: str) -> dict[str, Any]: + return { + "role": role, + "path": path.name, + "byte_length": path.stat().st_size, + "sha256": _sha256(path), + } + + +def _object(value: object, label: str) -> dict[str, Any]: + if not isinstance(value, dict): + raise E38PerceptionBaselineError(f"{label} must be an object") + return value + + +def _read_json(path: Path) -> dict[str, Any]: + try: + value = json.loads(path.read_text(encoding="utf-8-sig")) + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc: + raise E38PerceptionBaselineError(f"invalid JSON: {path.name}") from exc + return _object(value, path.name) + + +def _read_jsonl(path: Path) -> list[dict[str, Any]]: + rows: list[dict[str, Any]] = [] + try: + with path.open("r", encoding="utf-8-sig") as stream: + for line in stream: + rows.append(_object(json.loads(line), path.name)) + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc: + raise E38PerceptionBaselineError(f"invalid JSONL: {path.name}") from exc + return rows + + +def _write_json(path: Path, value: object) -> None: + with path.open("x", encoding="utf-8", newline="\n") as stream: + json.dump(value, stream, ensure_ascii=False, indent=2, sort_keys=True) + stream.write("\n") + stream.flush() + os.fsync(stream.fileno()) + + +def _write_jsonl(path: Path, rows: list[dict[str, Any]]) -> None: + with path.open("x", encoding="utf-8", newline="\n") as stream: + for row in rows: + stream.write( + json.dumps( + row, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ) + ) + stream.write("\n") + stream.flush() + os.fsync(stream.fileno()) + + +def _canonical_json(value: object) -> bytes: + return json.dumps( + value, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ).encode() + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + while chunk := stream.read(1024 * 1024): + digest.update(chunk) + return digest.hexdigest() + + +def _utc_now() -> str: + return datetime.now(UTC).isoformat(timespec="milliseconds").replace("+00:00", "Z") diff --git a/src/k1link/web/advanced_laboratory_api.py b/src/k1link/web/advanced_laboratory_api.py index 881f453..d67e43e 100644 --- a/src/k1link/web/advanced_laboratory_api.py +++ b/src/k1link/web/advanced_laboratory_api.py @@ -39,6 +39,11 @@ from k1link.compute.e37_acceptance_contract import ( E37AcceptanceContractError, read_e37_acceptance_contract, ) +from k1link.compute.e38_perception_baseline import ( + E38PerceptionBaseline, + E38PerceptionBaselineError, + read_e38_perception_baseline, +) LABORATORY_ADVANCED_CATALOG_SCHEMA: Final = ( "missioncore.laboratory-advanced-catalog/v1" @@ -50,6 +55,7 @@ _E33_RESULT_ID = re.compile(r"^e33-worker-shadow-[a-f0-9]{64}$") _E34_RESULT_ID = re.compile(r"^e34-temporal-occupied-[a-f0-9]{64}$") _E35_RESULT_ID = re.compile(r"^e35-degradation-recovery-[a-f0-9]{64}$") _E37_RESULT_ID = re.compile(r"^e37-ravnoves-acceptance-[a-f0-9]{64}$") +_E38_RESULT_ID = re.compile(r"^e38-perception-baseline-[a-f0-9]{64}$") RootProvider = Callable[[], Path | None] @@ -118,6 +124,15 @@ def _read_e37_cached( return read_e37_acceptance_contract(Path(root_text)) +@lru_cache(maxsize=16) +def _read_e38_cached( + root_text: str, + signature: tuple[int, ...], +) -> E38PerceptionBaseline: + del signature + return read_e38_perception_baseline(Path(root_text)) + + def _configured_root(provider: RootProvider) -> Path | None: value = provider() if value is None: @@ -555,6 +570,40 @@ def _project_e37(result: E37AcceptanceContract) -> dict[str, object]: } +def _project_e38(result: E38PerceptionBaseline) -> dict[str, object]: + identity = _object(result.manifest.get("identity"), "E38 identity") + source = _object(identity.get("source"), "E38 source") + execution = _object(identity.get("execution"), "E38 execution") + profile = _object(identity.get("profile"), "E38 profile") + metrics = _object(result.report.get("metrics"), "E38 metrics") + dimensions = _object(metrics.get("dimensions"), "E38 dimensions") + quality_gate = _object(result.report.get("quality_gate"), "E38 gate") + return { + "result_id": result.result_id, + "created_at_utc": result.manifest.get("created_at_utc"), + "source_session_id": source.get("session_id"), + "source_display_name": source.get("display_name"), + "status": result.report.get("status"), + "profile_id": profile.get("profile_id"), + "worker_node": execution.get("worker_node"), + "quality_gate_passed": quality_gate.get("passed"), + "metrics": { + "development_items": metrics.get("development_items"), + "validation_items": metrics.get("validation_items"), + "terminal_outcomes": metrics.get("terminal_outcomes"), + "accounting_fraction": metrics.get("accounting_fraction"), + "false_free_claims": metrics.get("false_free_claims"), + "high_severity_failures": metrics.get("high_severity_failures"), + "dimensions": copy.deepcopy(dimensions), + }, + "blocking_checks": copy.deepcopy(quality_gate.get("blocking_checks")), + "decision": copy.deepcopy(result.report.get("decision")), + "limitations": copy.deepcopy(result.report.get("limitations")), + "authority": copy.deepcopy(result.report.get("authority")), + "access": "read-only", + } + + def _empty_catalog(configured: bool) -> dict[str, object]: return { "schema_version": LABORATORY_ADVANCED_CATALOG_SCHEMA, @@ -574,6 +623,7 @@ def build_advanced_laboratory_router( e34_root_provider: RootProvider = lambda: None, e35_root_provider: RootProvider = lambda: None, e37_root_provider: RootProvider = lambda: None, + e38_root_provider: RootProvider = lambda: None, ) -> APIRouter: router = APIRouter(prefix="/api/v1/laboratory", tags=["laboratory"]) @@ -803,4 +853,37 @@ def build_advanced_laboratory_router( "invalid_total": invalid_total, } + @router.get("/e38/results") + def list_e38_results( + limit: int = Query(default=1, ge=1, le=10), + ) -> dict[str, object]: + root = _configured_root(e38_root_provider) + if root is None: + return _empty_catalog(False) + candidates = _candidates(root, _E38_RESULT_ID) + items: list[dict[str, object]] = [] + invalid_total = 0 + for candidate in candidates: + try: + result = _read_e38_cached( + str(candidate.resolve()), + _result_signature(candidate), + ) + if len(items) < limit: + items.append(_project_e38(result)) + except ( + E38PerceptionBaselineError, + KeyError, + OSError, + TypeError, + ValueError, + ): + invalid_total += 1 + return { + **_empty_catalog(True), + "items": items, + "candidate_total": len(candidates), + "invalid_total": invalid_total, + } + return router diff --git a/src/k1link/web/app.py b/src/k1link/web/app.py index 3ff6bc2..6afd421 100644 --- a/src/k1link/web/app.py +++ b/src/k1link/web/app.py @@ -543,6 +543,13 @@ app.include_router( / "e37" / "results" ), + e38_root_provider=lambda: ( + REPOSITORY_ROOT + / ".runtime" + / "compute-experiments" + / "e38" + / "results" + ), ) ) app.include_router( diff --git a/tests/test_advanced_laboratory_api.py b/tests/test_advanced_laboratory_api.py index 9d7964e..576dc44 100644 --- a/tests/test_advanced_laboratory_api.py +++ b/tests/test_advanced_laboratory_api.py @@ -26,7 +26,7 @@ def _endpoint(router: APIRouter, path: str) -> object: def test_advanced_catalogs_are_empty_when_not_configured() -> None: router = build_advanced_laboratory_router() - for name in ("e31", "e32", "e33", "e34", "e35", "e37"): + for name in ("e31", "e32", "e33", "e34", "e35", "e37", "e38"): route = _endpoint(router, f"/api/v1/laboratory/{name}/results") catalog = route(limit=1) # type: ignore[operator] assert catalog == { @@ -48,7 +48,8 @@ def test_advanced_catalogs_fail_closed_on_incomplete_results( e34 = tmp_path / "e34" e35 = tmp_path / "e35" e37 = tmp_path / "e37" - for root in (e31, e32, e33, e34, e35, e37): + e38 = tmp_path / "e38" + for root in (e31, e32, e33, e34, e35, e37, e38): root.mkdir() (e31 / f"e31-source-qualification-{'1' * 64}").mkdir() (e32 / f"e32-track-geometry-{'2' * 64}").mkdir() @@ -56,6 +57,7 @@ def test_advanced_catalogs_fail_closed_on_incomplete_results( (e34 / f"e34-temporal-occupied-{'4' * 64}").mkdir() (e35 / f"e35-degradation-recovery-{'5' * 64}").mkdir() (e37 / f"e37-ravnoves-acceptance-{'7' * 64}").mkdir() + (e38 / f"e38-perception-baseline-{'8' * 64}").mkdir() router = build_advanced_laboratory_router( e31_root_provider=lambda: e31, e32_root_provider=lambda: e32, @@ -63,9 +65,10 @@ def test_advanced_catalogs_fail_closed_on_incomplete_results( e34_root_provider=lambda: e34, e35_root_provider=lambda: e35, e37_root_provider=lambda: e37, + e38_root_provider=lambda: e38, ) - for name in ("e31", "e32", "e33", "e34", "e35", "e37"): + for name in ("e31", "e32", "e33", "e34", "e35", "e37", "e38"): route = _endpoint(router, f"/api/v1/laboratory/{name}/results") catalog = route(limit=1) # type: ignore[operator] assert catalog["configured"] is True @@ -422,3 +425,105 @@ def test_e37_catalog_projects_the_frozen_r0_contract( assert item["quality_target_evaluated"] is False assert item["authority"] == authority assert item["access"] == "read-only" + + +def test_e38_catalog_projects_failed_dimensions_without_hiding_result( + tmp_path: Path, + monkeypatch: MonkeyPatch, +) -> None: + result_id = f"e38-perception-baseline-{'8' * 64}" + root = tmp_path / "e38" + candidate = root / result_id + candidate.mkdir(parents=True) + (candidate / "manifest.json").write_text("{}", encoding="utf-8") + authority = { + "commands_enabled": False, + "navigation_or_safety_accepted": False, + } + dimension = { + "correct": 120, + "incorrect": 26, + "total": 146, + "accuracy": 0.821918, + "target": 0.9, + "passed": False, + "confusion": [], + "by_stratum": {}, + } + result = SimpleNamespace( + result_id=result_id, + manifest={ + "created_at_utc": "2026-07-28T02:50:00Z", + "identity": { + "source": { + "session_id": "20260720T065719Z_viewer_live", + "display_name": "RAVNOVES00", + }, + "profile": { + "profile_id": "e38-ravnoves00-r1-development-cart/v1", + }, + "execution": { + "worker_node": "DESKTOP-OPJ8J04", + }, + }, + }, + report={ + "status": "measured-r1-source-scoped-baseline", + "metrics": { + "development_items": 340, + "validation_items": 146, + "terminal_outcomes": 146, + "accounting_fraction": 1.0, + "false_free_claims": 0, + "high_severity_failures": 14, + "dimensions": { + "presence": dimension, + "geometry_association": dimension, + "freshness": { + **dimension, + "correct": 139, + "incorrect": 7, + "accuracy": 0.952055, + "passed": True, + }, + }, + }, + "quality_gate": { + "passed": False, + "blocking_checks": [ + "presence_target_reached", + "geometry_association_target_reached", + ], + }, + "decision": { + "r1_baseline_measured": True, + "accepted_for_release": False, + }, + "limitations": ["source-scoped"], + "authority": authority, + }, + ) + + def fake_read( + root_text: str, + signature: tuple[int, ...], + ) -> SimpleNamespace: + assert root_text == str(candidate.resolve()) + assert signature + return result + + monkeypatch.setattr(advanced_api, "_read_e38_cached", fake_read) + router = build_advanced_laboratory_router( + e38_root_provider=lambda: root, + ) + route = _endpoint(router, "/api/v1/laboratory/e38/results") + catalog = route(limit=1) # type: ignore[operator] + + assert catalog["candidate_total"] == 1 + assert catalog["invalid_total"] == 0 + item = catalog["items"][0] + assert item["quality_gate_passed"] is False + assert item["metrics"]["dimensions"]["freshness"]["passed"] is True + assert item["metrics"]["dimensions"]["presence"]["passed"] is False + assert item["authority"] == authority + assert item["access"] == "read-only" diff --git a/tests/test_e38_perception_baseline.py b/tests/test_e38_perception_baseline.py new file mode 100644 index 0000000..8fc6f14 --- /dev/null +++ b/tests/test_e38_perception_baseline.py @@ -0,0 +1,60 @@ +from __future__ import annotations + +from k1link.compute.e38_perception_baseline import ( + _dimension_metrics, + _predict_tree, + _train_tree, +) + + +def test_e38_cart_is_deterministic_and_respects_minimum_leaf() -> None: + rows = [ + ({"score": 0.1, "support": 0.0}, "background"), + ({"score": 0.2, "support": 1.0}, "background"), + ({"score": 0.7, "support": 3.0}, "object"), + ({"score": 0.8, "support": 4.0}, "object"), + ({"score": 0.9, "support": 5.0}, "object"), + ({"score": 1.0, "support": 6.0}, "object"), + ] + first = _train_tree( + rows, + feature_names=["score", "support"], + max_depth=3, + min_leaf=2, + ) + second = _train_tree( + rows, + feature_names=["score", "support"], + max_depth=3, + min_leaf=2, + ) + assert first == second + assert _predict_tree(first, {"score": 0.15, "support": 1.0}) == "background" + assert _predict_tree(first, {"score": 0.85, "support": 4.0}) == "object" + + +def test_e38_dimension_metrics_do_not_blend_strata() -> None: + rows = [ + { + "source_stratum": "agree", + "prediction": {"presence": "object-present"}, + "reference": {"presence": "object-present"}, + }, + { + "source_stratum": "agree", + "prediction": {"presence": "object-present"}, + "reference": {"presence": "background-or-noise"}, + }, + { + "source_stratum": "geometry-only", + "prediction": {"presence": "occupied-environment"}, + "reference": {"presence": "occupied-environment"}, + }, + ] + metrics = _dimension_metrics(rows, dimension="presence", target=0.9) + assert metrics["correct"] == 2 + assert metrics["incorrect"] == 1 + assert metrics["accuracy"] == 0.666667 + assert metrics["passed"] is False + assert metrics["by_stratum"]["agree"]["accuracy"] == 0.5 + assert metrics["by_stratum"]["geometry-only"]["accuracy"] == 1.0 diff --git a/tests/test_e38_worker_package.py b/tests/test_e38_worker_package.py new file mode 100644 index 0000000..1c8fc0e --- /dev/null +++ b/tests/test_e38_worker_package.py @@ -0,0 +1,86 @@ +from __future__ import annotations + +import importlib.util +import subprocess +import sys +from pathlib import Path + + +def _module() -> object: + path = ( + Path(__file__).resolve().parents[1] + / "experiments" + / "perception" + / "prepare_e38_worker_package.py" + ) + spec = importlib.util.spec_from_file_location("e38_worker_package_test", path) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + sys.modules[spec.name] = module + spec.loader.exec_module(module) + return module + + +def test_e38_package_is_minimal_content_addressed_projection(tmp_path: Path) -> None: + module = _module() + repository = Path(__file__).resolve().parents[1] + acceptance = ( + repository + / ".runtime" + / "compute-experiments" + / "e37" + / "results" + / ( + "e37-ravnoves-acceptance-" + "01b1efd586f747341c712d82f0907b39436a6f91ae92b1dfae987eca05fd8344" + ) + ) + materialization = ( + repository + / ".runtime" + / "compute-experiments" + / "e30" + / "materializations" + / ( + "e30-materialization-" + "841af926d8d28ab93538c46d8f31278a2234c4d1c12c7dc4dc296b249d59735a" + ) + ) + package = module.build_e38_worker_package( + repository_root=repository, + acceptance_root=acceptance, + materialization_root=materialization, + profile_path=( + repository + / "experiments" + / "perception" + / "e38_ravnoves00_r1_baseline_profile.json" + ), + output_root=tmp_path, + ) + manifest = module.validate_e38_worker_package(package) + assert package.name == f"e38-worker-package-{manifest['identity_sha256']}" + assert manifest["identity"]["classification"] == ( + "immutable-ravnoves00-r1-worker-input" + ) + assert len(manifest["artifacts"]) == 13 + completed = subprocess.run( + [ + sys.executable, + str(package / "runtime" / "run_e38_perception_baseline.py"), + "--package", + str(package), + "--output-root", + str(tmp_path / "results"), + ], + env={ + "PYTHONPATH": str(package / "runtime"), + "PYTHONDONTWRITEBYTECODE": "1", + "E38_WORKER_NODE": "TEST-WORKER-006", + }, + check=False, + capture_output=True, + text=True, + ) + assert completed.returncode == 0, completed.stderr + assert '"quality_gate_passed": false' in completed.stdout