diff --git a/.gitignore b/.gitignore
index 41cccf31..43faf39d 100644
--- a/.gitignore
+++ b/.gitignore
@@ -13,3 +13,5 @@ target
.codex/*
!.codex/agents/
!.codex/agents/*.toml
+tests/perf/.bench-dist
+tests/perf/.bench-app-dist
diff --git a/apps/web/src/bench-flag.d.ts b/apps/web/src/bench-flag.d.ts
new file mode 100644
index 00000000..7679ce20
--- /dev/null
+++ b/apps/web/src/bench-flag.d.ts
@@ -0,0 +1,12 @@
+/**
+ * Compile-time flag for the benchmark-only instrumentation seam (#335).
+ *
+ * The ordinary production build defines it `false` (apps/web/vite.config.ts), so
+ * every benchmark branch is removed by dead-code elimination before
+ * minification. The benchmark-mode application build in `tests/perf` defines it
+ * `true`.
+ *
+ * It is declared here so BOTH builds typecheck against the same product source:
+ * the seam lives in real application code, never in a copied implementation.
+ */
+declare const __DOCKERMAP_BENCH_ACCEPTANCE__: boolean;
diff --git a/apps/web/src/components/AppShell.tsx b/apps/web/src/components/AppShell.tsx
index 75ff01c3..f5ef3f33 100644
--- a/apps/web/src/components/AppShell.tsx
+++ b/apps/web/src/components/AppShell.tsx
@@ -13,6 +13,7 @@ import { AppContext } from "../context";
import Icon, { type IconName } from "./Icon";
import CommandPalette from "./CommandPalette";
import RouteFocusManager from "./RouteFocusManager";
+import { ModelAcceptanceStamp } from "../lib/performance/modelAcceptance";
import { StateDot, Tag } from "./primitives";
import { UNAVAILABLE_USER } from "../lib/identity";
@@ -275,6 +276,14 @@ export default function AppShell({ onBearerSignOut }: { onBearerSignOut: () => v
+ {/*
+ Benchmark-only acceptance stamp (#335). It renders nothing and has no
+ effect in the product build; the benchmark application build stamps the
+ accepted revision token in the same commit that renders the accepted
+ model, so the capture can attribute a DOM repaint to a revision.
+ */}
+
+
setCommandOpen(false)} model={model} />
);
diff --git a/apps/web/src/hooks/useSystemModel.ts b/apps/web/src/hooks/useSystemModel.ts
index 8fb9f24f..2e18f838 100644
--- a/apps/web/src/hooks/useSystemModel.ts
+++ b/apps/web/src/hooks/useSystemModel.ts
@@ -5,6 +5,7 @@ import { projectRuntimeMap } from "../lib/atlas/project";
import type { AtlasEnvelope } from "../lib/atlas/types";
import type { EvidenceMode, ModelProvenance } from "../lib/evidence";
import { modelProvenanceForMode } from "../lib/evidence";
+import { recordModelAcceptance, recordModelLayers, useDeliveredModel } from "../lib/performance/modelAcceptance";
import { useApiResource } from "./useApiResource";
export interface SystemModelState {
@@ -62,6 +63,15 @@ export function useSystemModel(refreshTick: number, evidenceMode: EvidenceMode |
const built = buildModel(snapshot.data, runtimeMap.data);
lastModel.current = built;
lastProvenance.current = snapshot.provenance;
+ // The acceptance seam (#335). This is the exact point at which a fetched
+ // resource/revision pair BECOMES the coherent model the UI renders, so it is
+ // where the benchmark's stage-6 clock starts. It is compile-time gated and
+ // carries only an opaque timestamp + revision token; see
+ // lib/performance/modelAcceptance.tsx.
+if (__DOCKERMAP_BENCH_ACCEPTANCE__) {
+ recordModelAcceptance(snapshot.data.modelRevision, runtimeMap.data.modelRevision);
+recordModelLayers(snapshot.data, runtimeMap.data, built);
+}
return built;
}, [snapshot.data, snapshot.generation, snapshot.provenance, runtimeMap.data, runtimeMap.generation, runtimeMap.provenance]);
@@ -103,10 +113,16 @@ export function useSystemModel(refreshTick: number, evidenceMode: EvidenceMode |
}, [snapshot.data, snapshot.generation, snapshot.provenance, runtimeMap.data, runtimeMap.generation, runtimeMap.provenance]);
return {
- model,
- atlas,
- findings,
- modelProvenance,
+ /**
+ * The publication seam (#335). In the product build `useDeliveredModel` is
+ * the identity function — no state, no effects, no behavioural difference.
+ * The benchmark application build defines the compile-time flag, which is
+ * where the artificial presentation-delay control withholds a newly accepted
+ * publication from the render tree. One publication (model + atlas +
+ * findings + provenance) is delayed as a unit so the render tree can never
+ * observe a split state.
+ */
+ ...useDeliveredModel({ model, atlas, findings, modelProvenance }, model?.modelRevision ?? null),
loading: snapshot.loading || runtimeMap.loading,
error: snapshot.error ?? runtimeMap.error
};
diff --git a/apps/web/src/lib/performance/modelAcceptance.tsx b/apps/web/src/lib/performance/modelAcceptance.tsx
new file mode 100644
index 00000000..af9f7c50
--- /dev/null
+++ b/apps/web/src/lib/performance/modelAcceptance.tsx
@@ -0,0 +1,197 @@
+/**
+ * Benchmark-only instrumentation seam for the time-to-answer baseline (#335).
+ *
+ * This is REAL production application code, not a copy: `useSystemModel` calls
+ * `recordModelAcceptance()` at the exact point where a freshly fetched
+ * resource/revision pair becomes the coherent model the UI renders, and routes
+ * that publication through `useDeliveredModel()`. Both are gated by the
+ * compile-time constant `__DOCKERMAP_BENCH_ACCEPTANCE__`:
+ *
+ * - production (`apps/web/vite.config.ts`) defines it `false`, so the whole
+ * module collapses to `return value` / `return null` and every benchmark
+ * branch, event identifier and delay mechanism is eliminated from the shipped
+ * bundle (`tests/perf/productionIsolation.test.mjs` inspects the artifact);
+ * - the benchmark-mode application build in `tests/perf` defines it `true`,
+ * which is how the capture observes the real acceptance seam and how the
+ * stage-6/7 independence control injects an artificial presentation delay
+ * *after* acceptance.
+ *
+ * The seam carries no product payload and no telemetry. It records an opaque
+ * timing event plus the model revision token that was accepted, in an in-memory
+ * sink on the page, and (benchmark build only) stamps that same opaque token on
+ * the document root so the capture can prove the DOM content it times belongs to
+ * the accepted revision's render. Nothing leaves the page; nothing is uploaded.
+ */
+import { useEffect, useLayoutEffect, useRef, useState, type ReactElement } from "react";
+import type { DockerSnapshot, RuntimeMap } from "@dockermap/contracts";
+import { summarize, type SystemModel } from "../model";
+
+/** One accepted coherent model: opaque timings only. */
+export interface ModelAcceptanceEvent {
+ /** Monotonic sequence number within this page. */
+ seq: number;
+ /** `performance.now()` at the instant the coherent model was accepted. */
+ at: number;
+ /** The opaque daemon model revision token the accepted model belongs to. */
+ revision: string;
+ /** The exact snapshot/runtime-map pair consumed by useSystemModel. */
+ snapshotRevision: string;
+ runtimeMapRevision: string;
+}
+
+/** Benchmark-only diagnostic payload; it is drained by the capture harness. */
+export interface ModelLayerDiagnostic {
+ snapshot_revision: string;
+ runtime_map_revision: string;
+ snapshot_offline_count: number;
+ runtime_map_relevant_state: { revision: string; offline_or_not_running_service_count: number; offline_or_not_running_container_count: number };
+ coherent_pair_accepted: { accepted: boolean; snapshot_revision: string; runtime_map_revision: string };
+ derived_model_offline_value: number;
+ story_offline_value_pre_render: number;
+ rendered_home_offline_value: number | null;
+ fixture_generation: number | null;
+ monotonic_timestamp: number;
+}
+
+declare global {
+interface Window {
+ /** Benchmark build only. Absent from the production bundle. */
+ __dockermapBenchAcceptanceSink?: ModelAcceptanceEvent[];
+ /** Benchmark build only: artificial presentation delay in ms (0/absent = off). */
+ __dockermapBenchRenderDelayMs?: number;
+/** Revision-targeted delay used by the ordinary benchmark capture. */
+__dockermapBenchRenderDelayTarget?: string;
+/** One-shot seam-isolation delay, armed before the next coherent acceptance. */
+__dockermapBenchDelayAfterNextAcceptance?: boolean;
+/** Benchmark-only timestamp at which the post-acceptance delay timer began. */
+__dockermapBenchDelayStartedAt?: number;
+ /** Benchmark build only. Drained synchronously by the capture harness. */
+ __dockermapBenchLayerSink?: ModelLayerDiagnostic[];
+ /** Benchmark build only. Set by the harness before it advances a fixture. */
+ __dockermapBenchFixtureGeneration?: number;
+ }
+}
+
+const sink: ModelAcceptanceEvent[] = [];
+const layerSink: ModelLayerDiagnostic[] = [];
+let sequence = 0;
+let lastAcceptedPair: string | null = null;
+
+/**
+ * The acceptance seam. Called from the real model publication path in
+ * `useSystemModel` at the moment `buildModel()` output becomes the model the UI
+ * uses — NOT from a DOM mutation, and NOT from a copied benchmark implementation.
+ *
+ * Duplicate calls for the same revision (a re-render recomputing the memo) are
+ * ignored, so one accepted revision produces exactly one event.
+ */
+export function recordModelAcceptance(snapshotRevision: string | null, runtimeMapRevision: string | null): void {
+ if (!__DOCKERMAP_BENCH_ACCEPTANCE__) return;
+ if (!snapshotRevision || snapshotRevision !== runtimeMapRevision) return;
+ const pair = `${snapshotRevision}\u0000${runtimeMapRevision}`;
+ if (pair === lastAcceptedPair) return;
+ lastAcceptedPair = pair;
+ sequence += 1;
+ sink.push({ seq: sequence, at: performance.now(), revision: snapshotRevision, snapshotRevision, runtimeMapRevision });
+ // Bounded: the sink keeps only a recent window on a long-lived page.
+ if (sink.length > 256) sink.splice(0, 128);
+ window.__dockermapBenchAcceptanceSink = sink;
+}
+
+/** Records the actual inputs and model value at the coherent-publication seam. */
+export function recordModelLayers(snapshot: DockerSnapshot, runtimeMap: RuntimeMap, model: SystemModel): void {
+ if (!__DOCKERMAP_BENCH_ACCEPTANCE__) return;
+ const runtimeStates = runtimeMap.nodes.filter((node) => /offline|stopped|dead|down|exited|not.running/i.test(`${node.status ?? ""} ${node.service?.status ?? ""}`));
+ const diagnostic: ModelLayerDiagnostic = {
+ snapshot_revision: snapshot.modelRevision,
+ runtime_map_revision: runtimeMap.modelRevision,
+ snapshot_offline_count: snapshot.containers.filter((container) => /offline|stopped|dead|down|exited|not.running/i.test(container.status)).length,
+ runtime_map_relevant_state: {
+ revision: runtimeMap.modelRevision,
+ offline_or_not_running_service_count: runtimeStates.filter((node) => node.service !== undefined && node.service !== null).length,
+ offline_or_not_running_container_count: runtimeStates.filter((node) => node.type === "container").length
+ },
+ coherent_pair_accepted: { accepted: snapshot.modelRevision === runtimeMap.modelRevision, snapshot_revision: snapshot.modelRevision, runtime_map_revision: runtimeMap.modelRevision },
+ derived_model_offline_value: summarize(model).offline,
+ story_offline_value_pre_render: summarize(model).offline,
+ rendered_home_offline_value: null,
+ fixture_generation: window.__dockermapBenchFixtureGeneration ?? null,
+ monotonic_timestamp: performance.now()
+ };
+ layerSink.push(diagnostic);
+ if (layerSink.length > 256) layerSink.splice(0, 128);
+ window.__dockermapBenchLayerSink = layerSink;
+}
+
+/**
+ * The publication seam. In the product build this is the identity function: no
+ * state, no effects, no observable difference. In the benchmark build it is the
+ * point where the artificial presentation delay control withholds a newly
+ * accepted publication from the render tree.
+ *
+ * The delay is keyed on the accepted revision, never on the surrounding
+ * publication object (which is recreated on every render) — keying on the object
+ * would reschedule the timer on every render instead of once per publication.
+ */
+export function useDeliveredModel(value: T, revision: string | null): T {
+if (!__DOCKERMAP_BENCH_ACCEPTANCE__) return value;
+return useDelayedPublication(value, revision, window.__dockermapBenchRenderDelayTarget ?? null, window.__dockermapBenchDelayAfterNextAcceptance === true);
+}
+
+/**
+ * Benchmark-only: stamps the accepted revision token on the document root in the
+ * same commit that renders the accepted model, so the capture can attribute a
+ * DOM repaint to a revision instead of assuming it.
+ */
+export function ModelAcceptanceStamp({ revision }: { revision: string | null }): ReactElement | null {
+ if (!__DOCKERMAP_BENCH_ACCEPTANCE__) return null;
+ return ;
+}
+
+function AcceptedRevisionStamp({ revision }: { revision: string | null }): null {
+useLayoutEffect(() => {
+ const root = document.documentElement;
+ if (revision && revision.length > 0) root.dataset.dockermapAcceptedRevision = revision;
+else delete root.dataset.dockermapAcceptedRevision;
+ // This runs in the commit containing Home. Read its displayed metric instead
+ // of deriving a value from the revision or fixture generation.
+ const metric = [...document.querySelectorAll(".metric")].find((element) =>
+ element.querySelector(".metric-label")?.textContent?.trim() === "Offline"
+ );
+ const displayed = metric?.querySelector(".metric-value")?.textContent?.trim();
+ const latest = layerSink[layerSink.length - 1];
+ if (latest && latest.snapshot_revision === revision && displayed !== undefined && displayed !== "") {
+ const parsed = Number(displayed);
+ latest.rendered_home_offline_value = Number.isFinite(parsed) ? parsed : null;
+ }
+}, [revision]);
+ return null;
+}
+
+function useDelayedPublication(value: T, revision: string | null, delayTarget: string | null, delayAfterNextAcceptance: boolean): T {
+const deliveredRevision = useRef(revision);
+const isNextAcceptance = delayAfterNextAcceptance && Boolean(revision) && revision !== deliveredRevision.current;
+const delayMs = revision === delayTarget || isNextAcceptance ? armedDelayMs() : 0;
+const latest = useRef(value);
+latest.current = value;
+const [delivered, setDelivered] = useState(value);
+useEffect(() => {
+if (delayMs <= 0) return undefined;
+// This flag is consumed only after recordModelAcceptance() ran in the same
+// render. It deliberately identifies no daemon publication or trigger.
+if (isNextAcceptance) window.__dockermapBenchDelayAfterNextAcceptance = false;
+if (isNextAcceptance) window.__dockermapBenchDelayStartedAt = performance.now();
+const timer = window.setTimeout(() => {
+deliveredRevision.current = revision;
+setDelivered(latest.current);
+}, delayMs);
+return () => window.clearTimeout(timer);
+}, [revision, delayMs, isNextAcceptance]);
+return delayMs > 0 ? delivered : value;
+}
+
+function armedDelayMs(): number {
+ if (typeof window === "undefined") return 0;
+ const raw = Number(window.__dockermapBenchRenderDelayMs ?? 0);
+ return Number.isFinite(raw) && raw > 0 ? raw : 0;
+}
diff --git a/apps/web/src/lib/performance/timeToAnswerEvidence.test.ts b/apps/web/src/lib/performance/timeToAnswerEvidence.test.ts
new file mode 100644
index 00000000..429af697
--- /dev/null
+++ b/apps/web/src/lib/performance/timeToAnswerEvidence.test.ts
@@ -0,0 +1,275 @@
+import { describe, expect, it } from "vitest";
+import {
+ TIME_TO_ANSWER_BASELINE,
+ TIME_TO_ANSWER_CONTROLLED_RUNS,
+TIME_TO_ANSWER_MATRIX,
+ TIME_TO_ANSWER_WARM_UP_METRICS,
+ TIME_TO_ANSWER_REFERENCE_FIXTURES,
+ TIME_TO_ANSWER_STAGES,
+ TIME_TO_ANSWER_WARMED_SAMPLES,
+ assertTimeToAnswerPromotion,
+ compatibleTimeToAnswerEnvironment,
+ derivedTimeToAnswerSummaries,
+ summarizeTimeToAnswerStage,
+ timeToAnswerLimit,
+ timeToAnswerP95,
+ validateTimeToAnswerEvidence,
+ withinTimeToAnswerPromotionLimit
+} from "./timeToAnswerEvidence";
+
+const environment: Record = {
+ runnerClass: "hearth-dedicated-x64",
+ cpuClass: "pinned-4-vcpu",
+ osImage: "ubuntu-24.04@sha256:fixture",
+ osKernel: "7.0.0-31-generic",
+ nodeRevision: "22.23.2",
+ rustRevision: "1.88.0",
+ dockerRevision: "29.0.0",
+ ssePollIntervalMs: "2000",
+ daemonBinarySha256: "eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee",
+ daemonBinaryBuild: "cargo-build-release-locked-p-dockermap-daemon",
+ cargoRevision: "cargo-1.88.0",
+ harnessRevision: "dddddddddddddddddddddddddddddddddddddddd",
+ browserEngine: "chromium",
+ browserRevision: "1234567",
+ browserFlags: ["--disable-background-networking"],
+ fontEnvironment: "Noto-Sans-1.0",
+ buildMode: "production",
+ fixtureRevision: "dockermap-v1/time-to-answer-fixtures-1",
+ sourceRevision: "candidate",
+methodologyVersion: "dockermap-v1/time-to-answer-methodology-4"
+};
+
+/**
+ * Deliberately loosely typed: every hostile case below mutates the raw JSON a
+ * benchmark job would emit, and the validator must reject it without the test
+ * needing a cast per mutation.
+ */
+function rawEvidence(): {
+ baseline: string;
+ environment: Record;
+records: { fixture: string; stage: string; measurementProtocol: string; sourceEvidenceFile: string; checkpointSha: string; runs: number[][] }[];
+} {
+ return {
+ baseline: TIME_TO_ANSWER_BASELINE,
+ environment,
+ records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }, record) => ({
+fixture,
+stage,
+ measurementProtocol: stage === "publicationToNodeObservationMs" ? "controlled-poll-phase" : "end-to-end",
+ sourceEvidenceFile: stage === "publicationToNodeObservationMs" ? "stage-five.raw.json" : "general.raw.json",
+checkpointSha: "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
+runs: Array.from({ length: TIME_TO_ANSWER_CONTROLLED_RUNS }, (_, run) =>
+ Array.from(
+ { length: TIME_TO_ANSWER_WARMED_SAMPLES },
+ (_, sample) => record * 100 + run * 10 + sample + 1
+ )
+ )
+ }))
+ };
+}
+
+describe("time-to-answer evidence contract", () => {
+ it("defines a closed stage matrix covering every acceptance bucket", () => {
+ expect(new Set(TIME_TO_ANSWER_STAGES.map((stage) => stage.id)).size).toBe(
+ TIME_TO_ANSWER_STAGES.length
+ );
+ expect([...new Set(TIME_TO_ANSWER_STAGES.map((stage) => stage.bucket))].sort()).toEqual([
+ "backend-collection",
+ "browser-model",
+ "rendering",
+ "search",
+ "transport-notification"
+ ]);
+ // The three reference sizes are measured for every stage.
+ for (const stage of TIME_TO_ANSWER_STAGES) {
+ for (const reference of ["reference-25", "reference-100", "reference-250"]) {
+ expect(stage.fixtures).toContain(reference);
+ }
+ }
+ // Every listed fixture is a declared fixture, and the matrix is the union
+ // of the per-stage lists with no duplicate pair.
+ const declared = new Set(TIME_TO_ANSWER_REFERENCE_FIXTURES.map((fixture) => fixture.name));
+ const expectedPairs = TIME_TO_ANSWER_STAGES.flatMap((stage) => stage.fixtures).length;
+ expect(TIME_TO_ANSWER_MATRIX.length).toBe(expectedPairs);
+ expect(new Set(TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => `${fixture}\u0000${stage}`)).size).toBe(
+ TIME_TO_ANSWER_MATRIX.length
+ );
+ for (const { fixture } of TIME_TO_ANSWER_MATRIX) expect(declared.has(fixture)).toBe(true);
+ // The four scenario fixtures exist and are actually exercised.
+ for (const scenario of [
+ "provider-only-revision-change",
+ "docker-topology-change",
+ "slow-bounded-compose-projection",
+ "unavailable-optional-provider"
+ ]) {
+ expect(declared.has(scenario)).toBe(true);
+ expect(TIME_TO_ANSWER_MATRIX.some(({ fixture }) => fixture === scenario)).toBe(true);
+ }
+ });
+
+it("documents what each stage proves and does not prove", () => {
+ for (const stage of TIME_TO_ANSWER_STAGES) {
+ expect(stage.measures.length).toBeGreaterThan(20);
+ expect(stage.doesNotProve.length).toBeGreaterThan(20);
+ }
+ });
+
+ it("recomputes summaries from raw samples instead of trusting supplied values", () => {
+ const evidence = validateTimeToAnswerEvidence(rawEvidence());
+ const summaries = derivedTimeToAnswerSummaries(evidence);
+ expect(TIME_TO_ANSWER_CONTROLLED_RUNS).toBe(3);
+ expect(TIME_TO_ANSWER_WARMED_SAMPLES).toBe(15);
+ expect(timeToAnswerP95(Array.from({ length: 15 }, (_, index) => index))).toBe(14);
+ const warmed = Array.from({ length: 15 }, (_, index) => index + 1);
+ expect(summarizeTimeToAnswerStage([warmed, warmed, warmed])).toEqual({
+ runP95Ms: [15, 15, 15],
+ medianOfThreeRunP95Ms: 15,
+ reviewedMs: 15,
+ reviewedAggregation: "median-of-three-run-p95"
+ });
+ const first = TIME_TO_ANSWER_MATRIX[0]!;
+ expect(summaries.get(`${first.fixture}\u0000${first.stage}`)).toEqual({
+ runP95Ms: [15, 25, 35],
+ medianOfThreeRunP95Ms: 25,
+ reviewedMs: 25,
+ reviewedAggregation: "median-of-three-run-p95"
+ });
+ });
+
+ it("fails closed on fabricated, incomplete or hostile artifacts", () => {
+ expect(() =>
+ validateTimeToAnswerEvidence({ ...rawEvidence(), summary: "fabricated" })
+ ).toThrow("closed baseline/environment/records schema");
+ expect(() =>
+ validateTimeToAnswerEvidence({
+ ...rawEvidence(),
+ baseline: "dockermap-v1/other-baseline"
+ })
+ ).toThrow("closed baseline/environment/records schema");
+ expect(() =>
+ validateTimeToAnswerEvidence({
+ ...rawEvidence(),
+ records: rawEvidence().records.slice(1)
+ })
+ ).toThrow("exact fixture × stage matrix");
+
+ const shortRun = rawEvidence();
+ shortRun.records[0]!.runs[0] = [1];
+ expect(() => validateTimeToAnswerEvidence(shortRun)).toThrow("exactly 15");
+
+ const twoRuns = rawEvidence();
+ twoRuns.records[0]!.runs = twoRuns.records[0]!.runs.slice(0, 2);
+ expect(() => validateTimeToAnswerEvidence(twoRuns)).toThrow("three raw runs");
+
+ const negative = rawEvidence();
+ negative.records[0]!.runs[0] = Array.from({ length: 15 }, () => -1);
+ expect(() => validateTimeToAnswerEvidence(negative)).toThrow("finite non-negative");
+
+ const notANumber = rawEvidence();
+ notANumber.records[0]!.runs[0] = Array.from({ length: 15 }, () => "fast" as unknown as number);
+ expect(() => validateTimeToAnswerEvidence(notANumber)).toThrow("numeric");
+
+ const unknownStage = rawEvidence();
+ unknownStage.records[0]!.stage = "vibesMs" as never;
+ expect(() => validateTimeToAnswerEvidence(unknownStage)).toThrow("unsafe or incomplete shape");
+
+const duplicated = rawEvidence();
+duplicated.records[1] = { ...duplicated.records[0]! };
+expect(() => validateTimeToAnswerEvidence(duplicated)).toThrow("duplicate or unsupported");
+
+const wrongProtocol = rawEvidence();
+wrongProtocol.records.find((record) => record.stage === "publicationToNodeObservationMs")!.measurementProtocol = "end-to-end";
+ expect(() => validateTimeToAnswerEvidence(wrongProtocol)).toThrow("wrong measurement protocol");
+
+ const stageSixControl = rawEvidence();
+ stageSixControl.records.find((record) => record.stage === "notificationToCoherentModelMs")!.measurementProtocol = "controlled-stage6-stage7-seam-isolation";
+ expect(() => validateTimeToAnswerEvidence(stageSixControl)).toThrow("wrong measurement protocol");
+
+const missingProvenance = rawEvidence();
+delete (missingProvenance.records[0] as Partial<(typeof missingProvenance.records)[number]>).checkpointSha;
+expect(() => validateTimeToAnswerEvidence(missingProvenance)).toThrow("unsafe or incomplete shape");
+
+const invalidCheckpoint = rawEvidence();
+invalidCheckpoint.records[0]!.checkpointSha = "not-a-checkpoint";
+expect(() => validateTimeToAnswerEvidence(invalidCheckpoint)).toThrow("provenance");
+
+ const undeclaredFixture = rawEvidence();
+ undeclaredFixture.records[0]!.fixture = "reference-1000";
+ expect(() => validateTimeToAnswerEvidence(undeclaredFixture)).toThrow(
+ "duplicate or unsupported"
+ );
+ });
+
+ it("rejects unsafe or arbitrary runner metadata and only compares equivalent pinned environments", () => {
+ expect(() =>
+ validateTimeToAnswerEvidence({
+ ...rawEvidence(),
+ environment: { ...environment, rawHostPath: "/private/host" }
+ })
+ ).toThrow("closed safe metadata fields");
+ expect(() =>
+ validateTimeToAnswerEvidence({
+ ...rawEvidence(),
+ environment: { ...environment, fontEnvironment: "font with spaces" }
+ })
+ ).toThrow("closed safe metadata fields");
+ expect(() =>
+ validateTimeToAnswerEvidence({
+ ...rawEvidence(),
+ environment: { ...environment, browserEngine: "webkit" }
+ })
+ ).toThrow("closed safe metadata fields");
+
+ const baseline = validateTimeToAnswerEvidence(rawEvidence()).environment;
+ expect(compatibleTimeToAnswerEnvironment(baseline, { ...baseline, sourceRevision: "other" })).toBe(
+ true
+ );
+ expect(compatibleTimeToAnswerEnvironment(baseline, { ...baseline, browserRevision: "9" })).toBe(
+ false
+ );
+ // dockerRevision is informational: no measured stage exercises the host
+ // Docker daemon, so a host engine upgrade must not invalidate a comparison.
+ expect(compatibleTimeToAnswerEnvironment(baseline, { ...baseline, dockerRevision: "30.0.0" })).toBe(
+ true
+ );
+ // A dimension this benchmark really pins still has to match.
+ expect(compatibleTimeToAnswerEnvironment(baseline, { ...baseline, osImage: "debian-13" })).toBe(
+ false
+ );
+ expect(compatibleTimeToAnswerEnvironment(baseline, { ...baseline, fixtureRevision: "v2" })).toBe(
+ false
+ );
+ });
+
+ it("uses the reviewed max(baseline × 1.25, baseline + 2 ms) promotion gate", () => {
+ expect(timeToAnswerLimit(4)).toBe(6);
+ expect(timeToAnswerLimit(20)).toBe(25);
+ expect(withinTimeToAnswerPromotionLimit(20, 25)).toBe(true);
+ expect(withinTimeToAnswerPromotionLimit(20, 25.01)).toBe(false);
+ expect(() => timeToAnswerLimit(-1)).toThrow("finite non-negative");
+ expect(() => timeToAnswerLimit(Number.NaN)).toThrow("finite non-negative");
+
+ const candidate = rawEvidence();
+ candidate.environment = { ...candidate.environment, sourceRevision: "other" };
+ expect(() => assertTimeToAnswerPromotion(rawEvidence(), candidate)).not.toThrow();
+
+ const slower = rawEvidence();
+ slower.records[0]!.runs = slower.records[0]!.runs.map((run) => run.map(() => 1_000_000));
+ expect(() => assertTimeToAnswerPromotion(rawEvidence(), slower)).toThrow(
+ "exceeds the reviewed promotion limit"
+ );
+
+ const wrongEnvironment = rawEvidence();
+ wrongEnvironment.environment = { ...wrongEnvironment.environment, osImage: "debian-13" };
+ expect(() => assertTimeToAnswerPromotion(rawEvidence(), wrongEnvironment)).toThrow(
+ "does not match the pinned baseline environment"
+ );
+ });
+});
+
+ it("derives end-to-end calibration ownership from baseline protocol ownership", () => {
+ expect(TIME_TO_ANSWER_WARM_UP_METRICS).not.toContain("publicationToNodeObservationMs");
+ expect(TIME_TO_ANSWER_WARM_UP_METRICS).toContain("notificationToCoherentModelMs");
+ expect(TIME_TO_ANSWER_WARM_UP_METRICS).toContain("coherentModelToUsefulRenderMs");
+ });
diff --git a/apps/web/src/lib/performance/timeToAnswerEvidence.ts b/apps/web/src/lib/performance/timeToAnswerEvidence.ts
new file mode 100644
index 00000000..a1328789
--- /dev/null
+++ b/apps/web/src/lib/performance/timeToAnswerEvidence.ts
@@ -0,0 +1,981 @@
+import { phaseMediansMs, phaseNormalizedP95Ms } from "./timeToAnswerPollPhase";
+
+/**
+ * DockerMap time-to-answer performance contract (#335).
+ *
+ * This module is the CLOSED schema and math for the controlled
+ * time-to-answer benchmark. It deliberately contains no timings and performs
+ * no measurement: the benchmark job runs on a pinned runner, writes a JSON
+ * artifact that stores RAW samples only, and this contract validates that
+ * artifact and derives every summary during review.
+ *
+ * Ordinary unit tests exercise this file's shape/math only. They must never
+ * compare elapsed time, be used as a performance gate, or be treated as
+ * evidence that DockerMap is fast.
+ */
+
+/**
+ * Closed list of the stages the benchmark measures, in the order the operator
+ * path experiences them. `measures` is the number's meaning; `doesNotProve`
+ * is the part a reader must not infer from it.
+ */
+export const TIME_TO_ANSWER_STAGES = [
+ {
+ id: "daemonStartToListenerMs",
+ bucket: "backend-collection",
+ measures: "Daemon process start until its HTTP listener accepts a request.",
+ doesNotProve:
+ "Nothing about collection. A fast listener with a slow first answer is still a slow product.",
+ fixtures: ["reference-25", "reference-100", "reference-250"]
+ },
+ {
+ id: "listenerToFirstDockerModelMs",
+ bucket: "backend-collection",
+ measures:
+ "Listener readiness until the first authoritative Docker model is observable to a reader.",
+ doesNotProve:
+ "Does not include anything the browser does, and does not prove the model is complete: optional provider evidence may still be absent.",
+ fixtures: ["reference-25", "reference-100", "reference-250"]
+ },
+ {
+ id: "dockerObservationMs",
+ bucket: "backend-collection",
+ measures: "One Docker inventory observation pass against the pinned fixture host.",
+ doesNotProve:
+ "Not a claim about a real Docker daemon's latency, host load, or image size; the fixture daemon is deterministic and local.",
+ fixtures: ["reference-25", "reference-100", "reference-250"]
+ },
+ {
+ id: "composeEnrichmentMs",
+ bucket: "backend-collection",
+ measures: "Compose filesystem correlation for one publication, timed separately from the Docker observation.",
+ doesNotProve:
+ "Not a claim about a real Compose project tree; and while the stages are still coupled this number is measured, not removed (see #336).",
+ fixtures: ["reference-25", "reference-100", "reference-250", "slow-bounded-compose-projection"]
+ },
+ {
+ id: "publicationToNodeObservationMs",
+ bucket: "transport-notification",
+ measures:
+ "Daemon publication until the Node/SSE layer observes that revision through TODAY'S real polling mechanism, poll wait included. The publication phase within the poll interval is DRIVEN, not hoped for: each recorded sample is assigned a declared phase on an explicit grid spanning the interval, the harness places the publication at that phase relative to its observation stream's poll ticks, and the observed phase is verified against the declared one before the sample is accepted.",
+ doesNotProve:
+ "Not browser work, not render, and not a claim about network distance to a remote operator. It is a phase response, not an observed user-traffic distribution: the reported phase-normalized summary weights the declared phases uniformly to characterise the latency the fixed polling mechanism imposes, and it does NOT claim that real host publications occur uniformly across poll phase.",
+ fixtures: [
+ "reference-25",
+ "reference-100",
+ "reference-250",
+ "provider-only-revision-change",
+ "docker-topology-change",
+ "unavailable-optional-provider"
+ ]
+ },
+ {
+ id: "notificationToCoherentModelMs",
+ bucket: "browser-model",
+ measures:
+ "Browser notification until the REAL application seam accepts one coherent model: the instant the fetched snapshot/runtime pair becomes the model the UI renders. It is observed at the application's own acceptance point, never derived from a DOM mutation.",
+ doesNotProve:
+ "Not a health judgement and not a statement that every evidence domain is current: it ends when a coherent model is accepted, not when the model is complete. It contains no rendering and says nothing about whether the operator saw anything.",
+ fixtures: [
+ "reference-25",
+ "reference-100",
+ "reference-250",
+ "provider-only-revision-change",
+ "docker-topology-change",
+ "unavailable-optional-provider"
+ ]
+ },
+ {
+ id: "coherentModelToUsefulRenderMs",
+ bucket: "rendering",
+ measures:
+ "From coherent-model acceptance until the accepted model's expected Home content is present — in a commit the application stamped with that accepted revision — followed by a bounded render/presentation confirmation (the probe observes the commit from an animation-frame loop and then awaits a bounded frame after it), and no sleeps. It shares no clock with the stage before it.",
+ doesNotProve:
+ "Not a visual-quality or accessibility claim, and not a claim that the operator found the answer. It is declared only for fixtures whose published change demonstrably repaints the Home content region; a provider-only or provider-unavailable revision is not guaranteed to repaint it, so measuring it there would be an empty number.",
+ fixtures: ["reference-25", "reference-100", "reference-250", "docker-topology-change"]
+ },
+ {
+ id: "buildModelMs",
+ bucket: "browser-model",
+ measures: "One `buildModel()` composition for the fixture model.",
+ doesNotProve: "Nothing about rendering, network, or findings derivation.",
+ fixtures: ["reference-25", "reference-100", "reference-250"]
+ },
+ {
+ id: "findingsDerivationMs",
+ bucket: "backend-collection",
+ measures:
+ "Findings derivation for the fixture's representative topology and evidence sizes. This runs in the daemon during publication, not in the browser.",
+ doesNotProve:
+ "Not a rule-quality claim, and it says nothing about a host with conditions the fixture does not contain. The fixture topology derives no findings, so this measures the empty-derivation path at its resolution floor.",
+ fixtures: ["reference-25", "reference-100", "reference-250"]
+ },
+ {
+ id: "legacyTopologyLayoutMs",
+ bucket: "rendering",
+ measures: "The legacy Home topology layout (force-layout preview) for the fixture model.",
+ doesNotProve:
+ "Not a claim about Atlas, and it does not by itself justify removing the preview; #338 decides that from this evidence.",
+ fixtures: ["reference-25", "reference-100", "reference-250"]
+ },
+ {
+ id: "commandQueryMs",
+ bucket: "search",
+ measures: "Cmd-K open plus query-to-results for the fixture's representative query classes.",
+ doesNotProve:
+ "Not a claim about answer quality, and the query set is a fixed representative sample, not operator behaviour.",
+ fixtures: ["reference-25", "reference-100", "reference-250"]
+ },
+ {
+ id: "productionBundleMs",
+ bucket: "rendering",
+ measures: "Production bundle/loading cost for the pinned production build.",
+ doesNotProve:
+ "Not a transfer-time claim for a real network; it is measured against the pinned local build and recorded in the pinned environment.",
+ fixtures: ["reference-25", "reference-100", "reference-250"]
+ }
+] as const;
+
+export type TimeToAnswerStage = (typeof TIME_TO_ANSWER_STAGES)[number];
+export type TimeToAnswerStageId = TimeToAnswerStage["id"];
+export type TimeToAnswerBucket = TimeToAnswerStage["bucket"];
+export type TimeToAnswerFixture = TimeToAnswerStage["fixtures"][number];
+
+export const TIME_TO_ANSWER_BASELINE = "dockermap-v1/time-to-answer-baseline-4";
+export const TIME_TO_ANSWER_WARMED_SAMPLES = 15;
+export const TIME_TO_ANSWER_CONTROLLED_RUNS = 3;
+
+/**
+ * The measurement design this contract describes. The baseline id names the
+ * CLOSED ARTIFACT SHAPE; the methodology version names HOW the numbers are
+ * produced — stage-5 deterministic phase control, the fixed burn-in policy, and
+ * the provenance/compatibility split. A candidate may
+ * only be compared against a baseline captured under the same methodology
+ * version, because a different design produces a different number for the same
+ * product.
+ */
+export const TIME_TO_ANSWER_METHODOLOGY = "dockermap-v1/time-to-answer-methodology-8";
+
+/**
+ * Every ordinary warmed end-to-end run executes exactly 60 fixed conditioning
+ * observations followed by 15 measured observations. Observation 61 is always
+ * the first measured sample. The conditioning observations are retained for
+ * audit and never enter timing summaries or promotion comparisons.
+ *
+ * This is deliberately a workload contract, not a steady-state claim: it does
+ * not infer stationarity, adapt to values, or guarantee that a metric settles.
+ */
+export const TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS = 60;
+/**
+ * Calibration is a separate, retained evidence exercise. Its constants are
+ * declared here (rather than in the runner) so a command cannot quietly tune
+ * them after it has seen an observation.
+ */
+// Fixed methodology-6 revision. Forty was selected after the previous
+// protocol could not validate its own late derived count; it is not a
+// statistically optimised or data-dependent window.
+export const TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS = 60;
+/** Historical calibration only; never Baseline-4 authority. */
+export const TIME_TO_ANSWER_WARM_UP_SAFETY_MARGIN = 2;
+
+/**
+ * Declared stationarity band: the median of the final two warm-up observations
+ * against the median of the measured window. This validity check is unchanged
+ * by the fixed-ten protocol revision.
+ */
+export const TIME_TO_ANSWER_STATIONARITY_MIN_RATIO = 0.5;
+export const TIME_TO_ANSWER_STATIONARITY_MAX_RATIO = 1.5;
+
+/**
+ * Historical calibration is retained as a rejected, non-authoritative audit
+ * diagnostic. Its per-metric results cannot select, alter, or invalidate the
+ * Baseline-4 fixed 60-observation burn-in.
+ */
+
+/** Reference fixtures (25/100/250 containers) plus the four scenario fixtures. */
+export const TIME_TO_ANSWER_REFERENCE_FIXTURES = [
+ { name: "reference-25", containers: 25, kind: "reference" },
+ { name: "reference-100", containers: 100, kind: "reference" },
+ { name: "reference-250", containers: 250, kind: "reference" },
+ { name: "provider-only-revision-change", containers: 100, kind: "scenario" },
+ { name: "docker-topology-change", containers: 100, kind: "scenario" },
+ { name: "slow-bounded-compose-projection", containers: 100, kind: "scenario" },
+ { name: "unavailable-optional-provider", containers: 100, kind: "scenario" }
+] as const;
+
+/** The closed fixture × stage matrix, derived from each stage's fixture list. */
+export const TIME_TO_ANSWER_MATRIX = TIME_TO_ANSWER_STAGES.flatMap((stage) =>
+stage.fixtures.map((fixture) => ({ fixture, stage: stage.id as TimeToAnswerStageId }))
+);
+
+/** Baseline-cell ownership is the authority for timing evidence producers. */
+export type TimeToAnswerBaselineProtocol = "controlled-poll-phase" | "end-to-end";
+export function timeToAnswerBaselineProtocol(stage: TimeToAnswerStageId): TimeToAnswerBaselineProtocol {
+ return stage === "publicationToNodeObservationMs" ? "controlled-poll-phase" : "end-to-end";
+}
+export const TIME_TO_ANSWER_END_TO_END_MATRIX = TIME_TO_ANSWER_MATRIX.filter(
+ (cell) => timeToAnswerBaselineProtocol(cell.stage) === "end-to-end"
+);
+export const TIME_TO_ANSWER_CONTROLLED_POLL_MATRIX = TIME_TO_ANSWER_MATRIX.filter(
+ (cell) => timeToAnswerBaselineProtocol(cell.stage) === "controlled-poll-phase"
+);
+
+/**
+ * Every pinned dimension of a controlled run. A missing field, or a candidate
+ * whose pinned environment differs from the baseline, invalidates the record;
+ * it never justifies retrying until a preferred duration appears.
+ */
+export type TimeToAnswerEnvironment = {
+ runnerClass: string;
+ cpuClass: string;
+ osImage: string;
+ osKernel: string;
+ nodeRevision: string;
+ rustRevision: string;
+ dockerRevision: string;
+ /**
+ * The effective `DOCKERMAP_SSE_INTERVAL_MS` the API ran with. Stage 5
+ * measures today's real publication-observation mechanism, poll wait
+ * included, so the interval is part of the pinned environment: a candidate
+ * that changed it has not been measured against the same mechanism.
+ */
+ ssePollIntervalMs: string;
+ /**
+ * The revision of the benchmark harness itself (latest commit touching
+ * `tests/perf` and the performance contract). A baseline is only reproducible
+ * if both the product and the harness that measured it are identified: a
+ * number produced by an uncommitted harness cannot be re-derived by anyone.
+ */
+ /** sha256 of the exact release daemon executable this capture ran. */
+ daemonBinarySha256: string;
+ /** The command and profile that produced that binary. */
+ daemonBinaryBuild: string;
+ cargoRevision: string;
+ harnessRevision: string;
+ browserEngine: "chromium";
+ browserRevision: string;
+ browserFlags: readonly string[];
+ fontEnvironment: string;
+ buildMode: "production";
+ fixtureRevision: string;
+ sourceRevision: string;
+ /** The measurement design (see TIME_TO_ANSWER_METHODOLOGY). Required to match. */
+ methodologyVersion: string;
+};
+
+export interface TimeToAnswerRecord {
+fixture: string;
+stage: TimeToAnswerStageId;
+/** Controlled sub-benchmarks own only the cells they explicitly name. */
+ measurementProtocol: "controlled-poll-phase" | "controlled-stage6-stage7-seam-isolation" | "end-to-end";
+/** Raw evidence section that produced this one cell. */
+sourceEvidenceFile: string;
+/** Committed source/harness checkpoint that produced this cell. */
+checkpointSha: string;
+/** Each inner array is one complete controlled run of warmed samples. */
+runs: readonly (readonly number[])[];
+}
+
+export interface TimeToAnswerEvidence {
+ baseline: typeof TIME_TO_ANSWER_BASELINE;
+ environment: TimeToAnswerEnvironment;
+ records: readonly TimeToAnswerRecord[];
+}
+
+export interface TimeToAnswerStageSummary {
+ runP95Ms: readonly number[];
+ medianOfThreeRunP95Ms: number;
+ /** The authority used for review and promotion of this record. */
+ reviewedMs: number;
+ reviewedAggregation: "median-of-three-run-p95" | "phase-normalized-p95";
+}
+
+const environmentKeys = [
+ "runnerClass",
+ "cpuClass",
+ "osImage",
+ "osKernel",
+ "nodeRevision",
+ "rustRevision",
+ "dockerRevision",
+ "ssePollIntervalMs",
+ "harnessRevision",
+ "daemonBinarySha256",
+ "daemonBinaryBuild",
+ "cargoRevision",
+ "browserEngine",
+ "browserRevision",
+ "browserFlags",
+ "fontEnvironment",
+ "buildMode",
+ "fixtureRevision",
+ "sourceRevision",
+ "methodologyVersion"
+] as const;
+const evidenceKeys = ["baseline", "environment", "records"] as const;
+const recordKeys = ["fixture", "stage", "measurementProtocol", "sourceEvidenceFile", "checkpointSha", "runs"] as const;
+const safeValue = /^[A-Za-z0-9._/@:+=-]{1,160}$/;
+const checkpointSha = /^[0-9a-f]{7,40}$/;
+const stageIds = new Set(TIME_TO_ANSWER_STAGES.map((stage) => stage.id));
+
+function isObject(value: unknown): value is Record {
+ return typeof value === "object" && value !== null && !Array.isArray(value);
+}
+
+function hasExactKeys(value: Record, keys: readonly string[]): boolean {
+ const actual = Object.keys(value).sort();
+ const expected = [...keys].sort();
+ return actual.length === expected.length && actual.every((key, index) => key === expected[index]);
+}
+
+function safeString(value: unknown): value is string {
+ return typeof value === "string" && safeValue.test(value);
+}
+
+/** Nearest-rank percentile: for 15 samples, p95 is the largest observed value. */
+export function timeToAnswerP95(samples: readonly number[]): number {
+ if (
+ samples.length !== TIME_TO_ANSWER_WARMED_SAMPLES ||
+ samples.some((sample) => typeof sample !== "number" || !Number.isFinite(sample) || sample < 0)
+ ) {
+ throw new Error(
+ `Time-to-answer benchmark requires exactly ${TIME_TO_ANSWER_WARMED_SAMPLES} finite non-negative warmed samples.`
+ );
+ }
+ const sorted = [...samples].sort((left, right) => left - right);
+ return sorted[Math.ceil(sorted.length * 0.95) - 1]!;
+}
+
+export function summarizeTimeToAnswerStage(
+ runs: readonly (readonly number[])[]
+): TimeToAnswerStageSummary {
+ if (runs.length !== TIME_TO_ANSWER_CONTROLLED_RUNS) {
+ throw new Error(
+ `Time-to-answer benchmark requires exactly ${TIME_TO_ANSWER_CONTROLLED_RUNS} complete controlled runs.`
+ );
+ }
+ const runP95Ms = runs.map(timeToAnswerP95);
+ const ordered = [...runP95Ms].sort((left, right) => left - right);
+ return {
+ runP95Ms,
+ medianOfThreeRunP95Ms: ordered[1]!,
+ reviewedMs: ordered[1]!,
+ reviewedAggregation: "median-of-three-run-p95"
+ };
+}
+
+export function assertTimeToAnswerEnvironment(
+ environment: unknown
+): asserts environment is TimeToAnswerEnvironment {
+ if (
+ !isObject(environment) ||
+ !hasExactKeys(environment, environmentKeys) ||
+ environment.browserEngine !== "chromium" ||
+ environment.buildMode !== "production" ||
+ ![
+ environment.runnerClass,
+ environment.cpuClass,
+ environment.osImage,
+ environment.osKernel,
+ environment.nodeRevision,
+ environment.rustRevision,
+ environment.dockerRevision,
+ environment.ssePollIntervalMs,
+ environment.harnessRevision,
+ environment.daemonBinarySha256,
+ environment.daemonBinaryBuild,
+ environment.cargoRevision,
+ environment.browserRevision,
+ environment.fontEnvironment,
+ environment.fixtureRevision,
+ environment.sourceRevision,
+ environment.methodologyVersion
+ ].every(safeString) ||
+ !Array.isArray(environment.browserFlags) ||
+ environment.browserFlags.length === 0 ||
+ environment.browserFlags.length > 16 ||
+ !environment.browserFlags.every(safeString)
+ ) {
+ throw new Error(
+ "Time-to-answer benchmark environment must use exactly the closed safe metadata fields for pinned runner/CPU/OS/kernel, Node/Rust/Docker revisions, Chromium revision and flags, fonts, production build, and fixture/source revision."
+ );
+ }
+}
+
+/**
+ * Reject untrusted JSON before deriving summaries. The artifact stores raw
+ * samples only; a supplied summary is not accepted as input.
+ */
+export function validateTimeToAnswerEvidence(value: unknown): TimeToAnswerEvidence {
+ if (
+ !isObject(value) ||
+ !hasExactKeys(value, evidenceKeys) ||
+ value.baseline !== TIME_TO_ANSWER_BASELINE ||
+ !Array.isArray(value.records)
+ ) {
+ throw new Error("Time-to-answer evidence must use the closed baseline/environment/records schema.");
+ }
+ assertTimeToAnswerEnvironment(value.environment);
+ const expected = new Set(TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => `${fixture}\u0000${stage}`));
+ if (value.records.length !== expected.size) {
+ throw new Error("Time-to-answer evidence must contain the exact fixture × stage matrix.");
+ }
+ const records = value.records.map((raw) => {
+ if (
+!isObject(raw) ||
+!hasExactKeys(raw, recordKeys) ||
+typeof raw.fixture !== "string" ||
+typeof raw.stage !== "string" ||
+!stageIds.has(raw.stage) ||
+typeof raw.measurementProtocol !== "string" ||
+typeof raw.sourceEvidenceFile !== "string" ||
+typeof raw.checkpointSha !== "string" ||
+!Array.isArray(raw.runs)
+ ) {
+ throw new Error("Time-to-answer record has an unsafe or incomplete shape.");
+ }
+ const key = `${raw.fixture}\u0000${raw.stage}`;
+if (!expected.delete(key)) {
+throw new Error("Time-to-answer evidence has a duplicate or unsupported fixture/stage record.");
+}
+ const requiredProtocol = timeToAnswerBaselineProtocol(raw.stage as TimeToAnswerStageId);
+if (raw.measurementProtocol !== requiredProtocol) {
+throw new Error(`Time-to-answer record ${raw.fixture}/${raw.stage} has the wrong measurement protocol.`);
+}
+if (!safeString(raw.sourceEvidenceFile) || !checkpointSha.test(raw.checkpointSha)) {
+throw new Error("Time-to-answer record provenance must contain safe source evidence and checkpoint identifiers.");
+}
+ if (
+ raw.runs.length !== TIME_TO_ANSWER_CONTROLLED_RUNS ||
+ !raw.runs.every((run) => Array.isArray(run))
+ ) {
+ throw new Error("Time-to-answer evidence requires exactly three raw runs per stage.");
+ }
+ const runs = raw.runs.map((run) =>
+ (run as unknown[]).map((sample) => {
+ if (typeof sample !== "number") throw new Error("Time-to-answer samples must be numeric.");
+ return sample;
+ })
+ );
+ // Executes the finite/non-negative/sample-count checks so summaries cannot
+ // be trusted input, and so a fabricated summary field cannot survive.
+ summarizeTimeToAnswerStage(runs);
+return {
+fixture: raw.fixture,
+stage: raw.stage as TimeToAnswerStageId,
+measurementProtocol: raw.measurementProtocol as TimeToAnswerRecord["measurementProtocol"],
+sourceEvidenceFile: raw.sourceEvidenceFile,
+checkpointSha: raw.checkpointSha,
+runs
+};
+ });
+ if (expected.size !== 0) {
+ throw new Error("Time-to-answer evidence is missing a required fixture/stage record.");
+ }
+ return { baseline: TIME_TO_ANSWER_BASELINE, environment: value.environment, records };
+}
+
+export function derivedTimeToAnswerSummaries(
+ evidence: TimeToAnswerEvidence
+): ReadonlyMap {
+ return new Map(
+ evidence.records.map((record) => {
+ const summary = summarizeTimeToAnswerStage(record.runs);
+if (record.measurementProtocol === "controlled-poll-phase") {
+ const normalized = derivedTimeToAnswerPhaseNormalized(record.runs, evidence.environment.ssePollIntervalMs);
+ return [
+ `${record.fixture}\u0000${record.stage}`,
+ {
+ ...summary,
+ reviewedMs: normalized.phaseNormalizedP95Ms,
+ reviewedAggregation: "phase-normalized-p95" as const
+ }
+ ];
+ }
+ return [`${record.fixture}\u0000${record.stage}`, summary];
+ })
+ );
+}
+
+/** Source revision deliberately differs between a baseline and its candidate. */
+/**
+ * What kind of measurement each stage is. The distinction is load-bearing, not
+ * descriptive: a warmed stage is a repeated steady-state operation whose first
+ * observation is a cold start, so that observation is recorded separately as
+ * warm-up and never enters the summary. A cold-start stage is the opposite —
+ * its first observation IS the measurement. Scenario-specific stages are only
+ * declared for fixtures that deliberately construct the scenario.
+ */
+export const TIME_TO_ANSWER_STAGE_KIND: Record = {
+ daemonStartToListenerMs: "cold-start",
+ listenerToFirstDockerModelMs: "cold-start",
+ dockerObservationMs: "warmed-repeated",
+ composeEnrichmentMs: "warmed-repeated",
+ publicationToNodeObservationMs: "warmed-repeated",
+ notificationToCoherentModelMs: "warmed-repeated",
+ coherentModelToUsefulRenderMs: "warmed-repeated",
+ buildModelMs: "warmed-repeated",
+ findingsDerivationMs: "warmed-repeated",
+ legacyTopologyLayoutMs: "warmed-repeated",
+ commandQueryMs: "warmed-repeated",
+ productionBundleMs: "warmed-repeated"
+};
+
+/** A stage measured on a scenario fixture is scenario-specific for that cell. */
+export function isScenarioCell(fixture: string, stage: string): boolean {
+ const declared = TIME_TO_ANSWER_REFERENCE_FIXTURES.find((entry) => entry.name === fixture);
+ return declared?.kind === "scenario" && TIME_TO_ANSWER_STAGE_KIND[stage] === "warmed-repeated";
+}
+
+export type WarmUpCalibrationCell = {
+ fixture: string;
+ metric: string;
+ observations: readonly number[];
+};
+
+export type WarmUpCalibrationDerivation = {
+ metric: string;
+ fixtureCounts: Readonly>;
+ frozenWarmUpCount: number;
+};
+
+export type WarmUpCalibrationCandidate = { candidate: number; ratio: number | null; inBand: boolean; sustained: boolean };
+export type WarmUpCalibrationFixtureReport = { fixture: string; stableWarmUpCount: number | null; candidates: readonly WarmUpCalibrationCandidate[]; reason: string | null };
+export type WarmUpCalibrationMetricReport = {
+ metric: string;
+ fixtures: readonly WarmUpCalibrationFixtureReport[];
+ maximumStableWarmUpCount: number | null;
+ proposedWarmUpCount: number | null;
+ requiredEvidenceLength: number | null;
+ evidenceBacked: boolean;
+ verdict: "PASS" | "CONFLICT";
+ reason: string | null;
+};
+export type WarmUpCalibrationReport = {
+ verdict: "PASS" | "CONFLICT";
+ metrics: readonly WarmUpCalibrationMetricReport[];
+ /** Historical diagnostic only; never Baseline-4 authority. */
+ nonAuthoritativeProposedWarmUpCounts: Readonly>;
+};
+
+/**
+ * The calibration population is closed independently of the baseline matrix:
+ * only the three size reference fixtures determine a warmed metric's count.
+ * Scenario cells are deliberately not a source of conditioning evidence.
+ */
+export const TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES = TIME_TO_ANSWER_REFERENCE_FIXTURES
+ .filter((fixture) => fixture.kind === "reference")
+ .map((fixture) => fixture.name);
+
+/** Every repeated end-to-end stage must be calibrated before baseline capture. */
+export const TIME_TO_ANSWER_WARM_UP_METRICS = TIME_TO_ANSWER_STAGES
+.filter((stage) =>
+ TIME_TO_ANSWER_STAGE_KIND[stage.id] === "warmed-repeated" &&
+ TIME_TO_ANSWER_END_TO_END_MATRIX.some((cell) => cell.stage === stage.id)
+)
+.map((stage) => stage.id);
+
+/** Median used by the existing stationarity semantics. */
+function median(values: readonly number[]): number {
+ if (values.length === 0) return Number.NaN;
+ const ordered = [...values].sort((left, right) => left - right);
+ const middle = Math.floor(ordered.length / 2);
+ return ordered.length % 2 === 0 ? (ordered[middle - 1]! + ordered[middle]!) / 2 : ordered[middle]!;
+}
+
+/**
+ * Derive one metric's frozen count from its complete fixed-window reference
+ * fixture cells. Candidate `w` compares obs[w-2:w] to obs[w:w+15]. The first
+ * candidate whose ratio stays inside the declared band at every later eligible
+ * position is selected for each fixture; the metric receives their maximum plus
+ * the fixed safety margin. The margin itself must still be evidence-backed.
+ */
+export function deriveFrozenWarmUpCount(cells: readonly WarmUpCalibrationCell[]): WarmUpCalibrationDerivation {
+ if (cells.length === 0) throw new Error("warm-up calibration needs at least one relevant reference fixture");
+ const metric = cells[0]!.metric;
+ if (cells.some((cell) => cell.metric !== metric)) throw new Error("warm-up calibration derives one metric at a time");
+ if (!TIME_TO_ANSWER_WARM_UP_METRICS.includes(metric as TimeToAnswerStageId)) {
+ throw new Error(`${metric} is not a warmed end-to-end metric`);
+ }
+ const expectedFixtures = new Set(TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES);
+ if (cells.length !== expectedFixtures.size) {
+ throw new Error(`${metric} calibration must retain every declared reference fixture`);
+ }
+ const fixtureCounts: Record = {};
+ const latestEligible = TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS - TIME_TO_ANSWER_WARMED_SAMPLES;
+ for (const cell of cells) {
+ if (!cell.fixture || fixtureCounts[cell.fixture] !== undefined || !expectedFixtures.delete(cell.fixture)) {
+ throw new Error("warm-up calibration fixtures must be the unique declared reference fixtures");
+ }
+ if (cell.observations.length !== TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS || cell.observations.some((value) => !Number.isFinite(value) || value < 0)) {
+ throw new Error(`${cell.fixture}/${metric} must retain exactly ${TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS} finite non-negative calibration observations`);
+ }
+ const stableAt = (candidate: number) => {
+ for (let position = candidate; position <= latestEligible; position += 1) {
+ const ratio = median(cell.observations.slice(position - 2, position)) / median(cell.observations.slice(position, position + TIME_TO_ANSWER_WARMED_SAMPLES));
+ if (!Number.isFinite(ratio) || ratio <= 0 || ratio < TIME_TO_ANSWER_STATIONARITY_MIN_RATIO || ratio > TIME_TO_ANSWER_STATIONARITY_MAX_RATIO) return false;
+ }
+ return true;
+ };
+ const earliest = Array.from({ length: latestEligible - 1 }, (_, index) => index + 2).find(stableAt);
+ if (earliest === undefined) throw new Error(`${cell.fixture}/${metric} never reaches sustained stationarity in the retained calibration window`);
+ fixtureCounts[cell.fixture] = earliest;
+ }
+ if (expectedFixtures.size !== 0) {
+ throw new Error(`${metric} calibration is missing declared reference fixtures: ${[...expectedFixtures].join(", ")}`);
+ }
+ const frozenWarmUpCount = Math.max(...Object.values(fixtureCounts)) + TIME_TO_ANSWER_WARM_UP_SAFETY_MARGIN;
+ if (frozenWarmUpCount > latestEligible) {
+ throw new Error(`${metric} calibration conflict: safety margin ${TIME_TO_ANSWER_WARM_UP_SAFETY_MARGIN} moves warm-up count ${frozenWarmUpCount} beyond the evidence-backed ${TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS}-observation window`);
+ }
+ return { metric, fixtureCounts, frozenWarmUpCount };
+}
+
+/** Derive every metric before returning an atomic overall calibration verdict. */
+export function deriveWarmUpCalibrationReport(cells: readonly WarmUpCalibrationCell[]): WarmUpCalibrationReport {
+ const latest = TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS - TIME_TO_ANSWER_WARMED_SAMPLES;
+ const metrics = TIME_TO_ANSWER_WARM_UP_METRICS.map((metric) => {
+ const metricCells = cells.filter((cell) => cell.metric === metric);
+ const fixtures = TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES.map((fixture) => {
+ const matches = metricCells.filter((cell) => cell.fixture === fixture);
+ const cell = matches.length === 1 ? matches[0] : undefined;
+ if (!cell || cell.observations.length !== TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS || cell.observations.some((value) => !Number.isFinite(value) || value < 0)) return { fixture, stableWarmUpCount: null, candidates: [], reason: !cell ? "missing reference fixture" : matches.length !== 1 ? "duplicate reference fixture" : `must retain exactly ${TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS} finite non-negative calibration observations` };
+ const candidates = Array.from({ length: latest - 1 }, (_, index) => index + 2).map((candidate) => {
+ const ratio = median(cell.observations.slice(candidate - 2, candidate)) / median(cell.observations.slice(candidate, candidate + TIME_TO_ANSWER_WARMED_SAMPLES));
+ return { candidate, ratio: Number.isFinite(ratio) ? ratio : null, inBand: Number.isFinite(ratio) && ratio > 0 && ratio >= TIME_TO_ANSWER_STATIONARITY_MIN_RATIO && ratio <= TIME_TO_ANSWER_STATIONARITY_MAX_RATIO, sustained: false };
+ });
+ const traced = candidates.map((candidate, index) => ({ ...candidate, sustained: candidate.inBand && candidates.slice(index).every((later) => later.inBand) }));
+ const stableWarmUpCount = traced.find((candidate) => candidate.sustained)?.candidate ?? null;
+ return { fixture, stableWarmUpCount, candidates: traced, reason: stableWarmUpCount === null ? "never reaches sustained stationarity in the retained calibration window" : null };
+ });
+ const expectedFixtures = new Set(TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES);
+ const unsupported = metricCells.some((cell) => !expectedFixtures.has(cell.fixture));
+ const counts = fixtures.map((fixture) => fixture.stableWarmUpCount);
+ const maximumStableWarmUpCount = counts.every((count): count is number => count !== null) ? Math.max(...counts) : null;
+ const proposedWarmUpCount = maximumStableWarmUpCount === null ? null : maximumStableWarmUpCount + TIME_TO_ANSWER_WARM_UP_SAFETY_MARGIN;
+ const requiredEvidenceLength = proposedWarmUpCount === null ? null : proposedWarmUpCount + TIME_TO_ANSWER_WARMED_SAMPLES;
+ const evidenceBacked = requiredEvidenceLength !== null && requiredEvidenceLength <= TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS;
+ const reason = unsupported ? "contains unsupported reference fixture" : fixtures.find((fixture) => fixture.reason)?.reason ?? (!evidenceBacked ? `safety margin requires ${requiredEvidenceLength} observations, exceeding the fixed ${TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS}-observation window` : null);
+ return { metric, fixtures, maximumStableWarmUpCount, proposedWarmUpCount, requiredEvidenceLength, evidenceBacked, verdict: reason ? "CONFLICT" as const : "PASS" as const, reason };
+ });
+ const passed = metrics.every((metric) => metric.verdict === "PASS");
+ return { verdict: passed ? "PASS" : "CONFLICT", metrics, nonAuthoritativeProposedWarmUpCounts: Object.freeze(passed ? Object.fromEntries(metrics.map((metric) => [metric.metric, metric.proposedWarmUpCount!])) : {}) };
+}
+
+/**
+ * Split one warmed measurement window into fixed burn-in
+ * observations and the recorded samples.
+ *
+ * The daemon's first-ever refresh runs before its listener binds, so its first
+ * passes through the collection path are cold. The count is FIXED by protocol
+ * (`TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS`), never chosen by looking at the data:
+ * with 15 recorded samples, nearest-rank p95 is the maximum, so a surviving cold
+ * observation would otherwise *become* the published number. Every warm-up
+ * observation is returned for the raw audit trail, and none of them enters the
+ * summary.
+ */
+export function splitWarmedObservations(
+observations: readonly number[],
+count = TIME_TO_ANSWER_WARMED_SAMPLES,
+ burnInCount = TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS
+): { warmUps: number[]; recorded: number[] } {
+ const required = count + burnInCount;
+ if (observations.length < required) {
+ throw new Error(
+ `a warmed stage needs at least ${required} observations: ${burnInCount} fixed ` +
+ `burn-in observations plus ${count} recorded samples`
+ );
+ }
+ const warmUps = observations.slice(0, burnInCount);
+ if (
+ [...warmUps, ...observations.slice(0, required)].some(
+ (value) => typeof value !== "number" || !Number.isFinite(value) || value < 0
+ )
+ ) {
+ throw new Error("warm-up and recorded observations must be finite non-negative numbers");
+ }
+ return { warmUps: [...warmUps], recorded: observations.slice(burnInCount, required) as number[] };
+}
+
+/**
+ * Declared stationarity check for one warmed cell/run. Compares the FINAL
+ * warm-up observations against the measured window using the predeclared band,
+ * and returns the ratio for the audit trail. A window whose warm-ups have not
+ * settled is INVALID — it is never repaired by discarding further samples,
+ * because choosing how many samples to drop after seeing the values would turn
+ * benchmark conditioning into result selection.
+ */
+export function assertWarmUpStationarity(input: {
+ label: string;
+warmUps: readonly number[];
+recorded: readonly number[];
+ warmUpCount: number;
+}): number {
+ const { label, warmUps, recorded, warmUpCount } = input;
+ if (warmUps.length !== warmUpCount) {
+ throw new Error(`${label} must retain exactly ${warmUpCount} warm-up observations`);
+ }
+ if (recorded.length !== TIME_TO_ANSWER_WARMED_SAMPLES) {
+ throw new Error(`${label} must record exactly ${TIME_TO_ANSWER_WARMED_SAMPLES} measured samples`);
+ }
+ const ratio = warmUpStationarityCalculation(warmUps, recorded).ratio;
+ if (!Number.isFinite(ratio) || ratio <= 0) {
+ throw new Error(`${label} has no usable warm-up/measured ratio`);
+ }
+ if (ratio > TIME_TO_ANSWER_STATIONARITY_MAX_RATIO || ratio < TIME_TO_ANSWER_STATIONARITY_MIN_RATIO) {
+ throw new Error(
+ `${label} is not stationary: the final warm-up observations sit at ${ratio.toFixed(2)}x the measured ` +
+ `median, outside the declared ${TIME_TO_ANSWER_STATIONARITY_MIN_RATIO}–` +
+ `${TIME_TO_ANSWER_STATIONARITY_MAX_RATIO}x band`
+ );
+ }
+ return ratio;
+}
+
+/** The audit calculation used by the unchanged warm-up stationarity validity check. */
+export function warmUpStationarityCalculation(
+ warmUps: readonly number[],
+ recorded: readonly number[]
+): { finalWarmUpMedian: number; measuredMedian: number; ratio: number } {
+ const finalWarmUpMedian = median(warmUps.slice(-2));
+ const measuredMedian = median(recorded);
+ return { finalWarmUpMedian, measuredMedian, ratio: finalWarmUpMedian / measuredMedian };
+}
+
+/**
+ * Bind the executed daemon binary to the recorded source revision. The benchmark
+ * does not claim bit-for-bit reproducible Rust builds across machines; it proves
+ * which binary THIS capture executed.
+ */
+export function assertDaemonBinaryProvenance(input: {
+ expectedSha256: string;
+ observedSha256: string;
+ phase: string;
+}): void {
+ if (!/^[0-9a-f]{64}$/.test(input.expectedSha256) || !/^[0-9a-f]{64}$/.test(input.observedSha256)) {
+ throw new Error("daemon binary provenance requires two lowercase sha256 digests");
+ }
+ if (input.expectedSha256 !== input.observedSha256) {
+ throw new Error(
+ `daemon binary provenance failed ${input.phase}: the executable is not the binary this capture pinned`
+ );
+ }
+}
+
+/**
+ * Provenance/identity keys: recorded so a baseline identifies exactly what was
+ * measured, but never a comparison REQUIREMENT. The executable digest and the
+ * source/harness revisions differ by construction for any legitimate candidate
+ * that changes the product or the harness, so requiring them to match would make
+ * comparison impossible — and the daemon binary is rebuilt from the candidate
+ * checkout, so a byte-identical digest is not even reproducible across a changed
+ * `CARGO_HOME`.
+ */
+export const TIME_TO_ANSWER_PROVENANCE_KEYS = [
+ "sourceRevision",
+ "harnessRevision",
+ "daemonBinarySha256"
+] as const;
+
+/**
+ * Recorded but informational: no measured stage exercises the host Docker daemon
+ * (the capture runs against the deterministic fixture daemon), so requiring this
+ * to match would fail a comparison for a dimension this benchmark never touches.
+ */
+export const TIME_TO_ANSWER_INFORMATIONAL_KEYS = ["dockerRevision"] as const;
+
+/**
+ * The keys a candidate must share with the baseline to be comparable at all:
+ * runner/CPU/OS/kernel, Node/Rust toolchain, cargo, browser engine/revision/
+ * flags, fonts, production build mode, fixture revision, the polling
+ * configuration, the daemon build command, and the benchmark methodology
+ * version. A time-to-answer number is only comparable to another number produced
+ * by the same design in the same environment.
+ */
+export const timeToAnswerCompatibilityKeys = environmentKeys.filter(
+ (key) =>
+ !(TIME_TO_ANSWER_PROVENANCE_KEYS as readonly string[]).includes(key) &&
+ !(TIME_TO_ANSWER_INFORMATIONAL_KEYS as readonly string[]).includes(key)
+);
+
+export function compatibleTimeToAnswerEnvironment(
+ baseline: TimeToAnswerEnvironment,
+ candidate: TimeToAnswerEnvironment
+): boolean {
+ return timeToAnswerCompatibilityKeys.every(
+ (key) => JSON.stringify(baseline[key]) === JSON.stringify(candidate[key])
+ );
+}
+
+/**
+ * The phase-normalized stage-5 figure (methodology revision 2): the observed
+ * latency median at each DECLARED phase, then nearest-rank p95 over those phase
+ * medians. Uniform weighting over the declared grid is a statement about the
+ * polling MECHANISM — never about real host publication phase or network
+ * distance.
+ */
+export function derivedTimeToAnswerPhaseNormalized(
+ runs: readonly (readonly number[])[],
+ ssePollIntervalMs: string
+): { phaseMediansMs: readonly number[]; phaseNormalizedP95Ms: number } {
+ const intervalMs = Number(ssePollIntervalMs);
+ if (!Number.isFinite(intervalMs) || intervalMs <= 0) {
+ throw new Error("the pinned SSE poll interval must be a positive number of milliseconds");
+ }
+ if (runs.length !== TIME_TO_ANSWER_CONTROLLED_RUNS) {
+ throw new Error(`stage 5 requires exactly ${TIME_TO_ANSWER_CONTROLLED_RUNS} controlled runs to normalise by phase`);
+ }
+ return {
+ phaseMediansMs: phaseMediansMs(runs, intervalMs),
+ phaseNormalizedP95Ms: phaseNormalizedP95Ms(runs, intervalMs)
+ };
+}
+
+/**
+ * Regression limits are derived from the measured baseline with the same
+ * reviewed rule the Atlas evidence uses; no aspirational absolute millisecond
+ * budget is invented before a baseline exists.
+ */
+export function timeToAnswerLimit(baselineMs: number): number {
+ if (!Number.isFinite(baselineMs) || baselineMs < 0) {
+ throw new Error("Time-to-answer baseline must be a finite non-negative duration.");
+ }
+ return Math.max(baselineMs * 1.25, baselineMs + 2);
+}
+
+export function withinTimeToAnswerPromotionLimit(baselineMs: number, candidateMs: number): boolean {
+ return Number.isFinite(candidateMs) && candidateMs >= 0 && candidateMs <= timeToAnswerLimit(baselineMs);
+}
+
+/** The benchmark job calls this after reading two closed JSON artifacts. */
+export function assertTimeToAnswerPromotion(baselineRaw: unknown, candidateRaw: unknown): void {
+ const baseline = validateTimeToAnswerEvidence(baselineRaw);
+ const candidate = validateTimeToAnswerEvidence(candidateRaw);
+ if (!compatibleTimeToAnswerEnvironment(baseline.environment, candidate.environment)) {
+ throw new Error("Time-to-answer candidate does not match the pinned baseline environment.");
+ }
+ const baselineSummaries = derivedTimeToAnswerSummaries(baseline);
+ for (const [key, candidateSummary] of derivedTimeToAnswerSummaries(candidate)) {
+ const baselineSummary = baselineSummaries.get(key);
+ if (
+ !baselineSummary ||
+ !withinTimeToAnswerPromotionLimit(
+ baselineSummary.reviewedMs,
+ candidateSummary.reviewedMs
+ )
+ ) {
+ throw new Error(`Time-to-answer candidate exceeds the reviewed promotion limit for ${key}.`);
+ }
+ }
+}
+
+/* ------------------------------------------------------------------ *
+ * Stage 6 / stage 7 independence control (#335)
+ *
+ * Stage 6 ends when the APPLICATION accepts a coherent model; stage 7 begins
+ * at that instant and ends when the accepted model's expected Home content has
+ * rendered (and one bounded frame has confirmed presentation). If the two
+ * numbers came from one clock, an artificial presentation delay injected AFTER
+ * acceptance would move both. The control therefore arms exactly that delay and
+ * requires stage 6 to stay put while stage 7 grows by the injected amount.
+ * ------------------------------------------------------------------ */
+
+/** The artificial presentation delay injected after coherent-model acceptance. */
+export const TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS = 250;
+/** Control samples per fixture that declares stages 6 and 7. */
+export const TIME_TO_ANSWER_INDEPENDENCE_SAMPLES = 3;
+/**
+ * Stage 6 must not move more than this. The allowance is generous relative to
+ * the delay: it exists to absorb ordinary run-to-run variance in a number that
+ * the control cannot legitimately affect, not to permit a shared clock.
+ */
+export const TIME_TO_ANSWER_INDEPENDENCE_STAGE_SIX_TOLERANCE_MS = 30;
+/** Stage 7 must absorb at least this share of the injected delay. */
+export const TIME_TO_ANSWER_INDEPENDENCE_STAGE_SEVEN_SHARE = 0.7;
+export interface StageSixSevenIndependence {
+ fixture: string;
+ delayMs: number;
+ normalStageSixMs: readonly number[];
+ normalStageSevenMs: readonly number[];
+ controlStageSixMs: readonly number[];
+ controlStageSevenMs: readonly number[];
+ stageSixMedianMs: number;
+ stageSixControlMedianMs: number;
+ stageSevenMedianMs: number;
+ stageSevenControlMedianMs: number;
+ stageSixDeltaMs: number;
+ stageSevenDeltaMs: number;
+}
+
+function assertSampleSet(label: string, values: readonly number[]): void {
+ if (values.length === 0) throw new Error(`${label} requires at least one sample`);
+ if (values.some((value) => typeof value !== "number" || !Number.isFinite(value) || value < 0)) {
+ throw new Error(`${label} requires finite non-negative samples`);
+ }
+}
+
+/**
+ * Enforce the independence control. Throws when the injected presentation delay
+ * fails to move stage 7 (the seam is measuring something other than
+ * presentation) or when it also moves stage 6 (both stages share a clock). A
+ * capture that cannot demonstrate this must not produce a baseline.
+ */
+export function assertStageSixSevenIndependence(input: {
+ fixture: string;
+ delayMs?: number;
+ normalStageSixMs: readonly number[];
+ normalStageSevenMs: readonly number[];
+ controlStageSixMs: readonly number[];
+ controlStageSevenMs: readonly number[];
+}): StageSixSevenIndependence {
+ const delayMs = input.delayMs ?? TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS;
+ if (!Number.isFinite(delayMs) || delayMs <= 0) {
+ throw new Error("the independence control requires a positive injected delay");
+ }
+ assertSampleSet("stage 6 normal samples", input.normalStageSixMs);
+ assertSampleSet("stage 7 normal samples", input.normalStageSevenMs);
+ assertSampleSet("stage 6 control samples", input.controlStageSixMs);
+ assertSampleSet("stage 7 control samples", input.controlStageSevenMs);
+
+ const stageSixMedianMs = median(input.normalStageSixMs);
+ const stageSixControlMedianMs = median(input.controlStageSixMs);
+ const stageSevenMedianMs = median(input.normalStageSevenMs);
+ const stageSevenControlMedianMs = median(input.controlStageSevenMs);
+ const stageSixDeltaMs = stageSixControlMedianMs - stageSixMedianMs;
+ const stageSevenDeltaMs = stageSevenControlMedianMs - stageSevenMedianMs;
+
+ const stageSixAllowance = Math.max(
+ TIME_TO_ANSWER_INDEPENDENCE_STAGE_SIX_TOLERANCE_MS,
+ stageSixMedianMs * 0.25
+ );
+ if (stageSixDeltaMs > stageSixAllowance) {
+ throw new Error(
+ `stage 6 moved by ${stageSixDeltaMs.toFixed(1)} ms under an artificial delay injected AFTER acceptance ` +
+ `(allowance ${stageSixAllowance.toFixed(1)} ms): stage 6 is not independent of presentation`
+ );
+ }
+ const requiredStageSevenDelta = delayMs * TIME_TO_ANSWER_INDEPENDENCE_STAGE_SEVEN_SHARE;
+ if (stageSevenDeltaMs < requiredStageSevenDelta) {
+ throw new Error(
+ `stage 7 only moved by ${stageSevenDeltaMs.toFixed(1)} ms for a ${delayMs} ms artificial delay ` +
+ `(required at least ${requiredStageSevenDelta.toFixed(1)} ms): stage 7 does not measure presentation of the accepted model`
+ );
+ }
+ if (input.controlStageSevenMs.some((value) => value < delayMs)) {
+ throw new Error("a control stage-7 sample is shorter than the injected delay, so the delay was not applied");
+ }
+ return {
+ fixture: input.fixture,
+ delayMs,
+ normalStageSixMs: input.normalStageSixMs,
+ normalStageSevenMs: input.normalStageSevenMs,
+ controlStageSixMs: input.controlStageSixMs,
+ controlStageSevenMs: input.controlStageSevenMs,
+ stageSixMedianMs,
+ stageSixControlMedianMs,
+ stageSevenMedianMs,
+ stageSevenControlMedianMs,
+ stageSixDeltaMs,
+ stageSevenDeltaMs
+ };
+}
diff --git a/apps/web/src/lib/performance/timeToAnswerIndependence.test.ts b/apps/web/src/lib/performance/timeToAnswerIndependence.test.ts
new file mode 100644
index 00000000..b85d44a2
--- /dev/null
+++ b/apps/web/src/lib/performance/timeToAnswerIndependence.test.ts
@@ -0,0 +1,105 @@
+import { describe, expect, it } from "vitest";
+import {
+ TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS,
+ assertStageSixSevenIndependence
+} from "./timeToAnswerEvidence";
+
+/**
+ * Stage 6/7 independence control (#335).
+ *
+ * Stage 6 ends when the application accepts a coherent model; stage 7 begins at
+ * that instant and ends when the accepted model's expected Home content has
+ * rendered. These tests pin the decision rule the capture enforces before it may
+ * produce a baseline: an artificial presentation delay injected AFTER acceptance
+ * must leave stage 6 alone and must move stage 7 by the injected amount.
+ */
+const NORMAL_STAGE_SIX = [12.4, 13.1, 11.8, 12.9, 12.2, 13.4, 12.0, 12.7, 13.0, 12.5];
+const NORMAL_STAGE_SEVEN = [4.1, 5.2, 3.8, 4.6, 5.0, 4.2, 3.9, 4.8, 4.4, 4.7];
+
+const control = (offsetMs: number) => NORMAL_STAGE_SEVEN.map((value) => value + offsetMs);
+
+describe("stage 6/7 independence control", () => {
+ it("accepts a control whose injected delay moves stage 7 and leaves stage 6 alone", () => {
+ const verdict = assertStageSixSevenIndependence({
+ fixture: "reference-25",
+ normalStageSixMs: NORMAL_STAGE_SIX,
+ normalStageSevenMs: NORMAL_STAGE_SEVEN,
+ controlStageSixMs: NORMAL_STAGE_SIX.map((value) => value + 2),
+ controlStageSevenMs: control(TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS)
+ });
+ expect(verdict.stageSevenDeltaMs).toBeCloseTo(TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS, 5);
+ expect(verdict.stageSixDeltaMs).toBeLessThanOrEqual(30);
+ });
+
+ it("rejects a seam whose stage 7 ignores the delayed presentation", () => {
+ // The RED case: a delayed render that does not move stage 7 means stage 7 is
+ // not measuring presentation of the accepted model (for example it is the
+ // same clock as stage 6, or it ends on the notification).
+ expect(() =>
+ assertStageSixSevenIndependence({
+ fixture: "reference-25",
+ normalStageSixMs: NORMAL_STAGE_SIX,
+ normalStageSevenMs: NORMAL_STAGE_SEVEN,
+ controlStageSixMs: NORMAL_STAGE_SIX,
+ controlStageSevenMs: NORMAL_STAGE_SEVEN
+ })
+ ).toThrow("stage 7 only moved by");
+ });
+
+ it("rejects a seam whose stage 6 moves with the delayed presentation", () => {
+ expect(() =>
+ assertStageSixSevenIndependence({
+ fixture: "reference-25",
+ normalStageSixMs: NORMAL_STAGE_SIX,
+ normalStageSevenMs: NORMAL_STAGE_SEVEN,
+ controlStageSixMs: NORMAL_STAGE_SIX.map((value) => value + 200),
+ controlStageSevenMs: control(TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS)
+ })
+ ).toThrow("stage 6 is not independent of presentation");
+ });
+
+ it("rejects a control where the delay was not actually applied", () => {
+ // Stage 7 moved slightly (noise) but a sample is still shorter than the
+ // injected delay, which can only mean the delay never reached the page.
+ expect(() =>
+ assertStageSixSevenIndependence({
+ fixture: "docker-topology-change",
+ normalStageSixMs: NORMAL_STAGE_SIX,
+ normalStageSevenMs: NORMAL_STAGE_SEVEN,
+ controlStageSixMs: NORMAL_STAGE_SIX,
+ controlStageSevenMs: [220, 300, 310]
+ })
+ ).toThrow("shorter than the injected delay");
+ });
+
+ it("rejects a partial delay absorption below the reviewed share", () => {
+ expect(() =>
+ assertStageSixSevenIndependence({
+ fixture: "reference-100",
+ normalStageSixMs: NORMAL_STAGE_SIX,
+ normalStageSevenMs: NORMAL_STAGE_SEVEN,
+ controlStageSixMs: NORMAL_STAGE_SIX,
+ controlStageSevenMs: control(TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS * 0.5)
+ })
+ ).toThrow("stage 7 only moved by");
+ });
+
+ it("refuses to judge an empty, malformed or delay-free control", () => {
+ const base = {
+ fixture: "reference-250",
+ normalStageSixMs: NORMAL_STAGE_SIX,
+ normalStageSevenMs: NORMAL_STAGE_SEVEN,
+ controlStageSixMs: NORMAL_STAGE_SIX,
+ controlStageSevenMs: control(TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS)
+ };
+ expect(() => assertStageSixSevenIndependence({ ...base, controlStageSevenMs: [] })).toThrow(
+ "requires at least one sample"
+ );
+ expect(() =>
+ assertStageSixSevenIndependence({ ...base, controlStageSixMs: [Number.NaN] })
+ ).toThrow("finite non-negative samples");
+ expect(() => assertStageSixSevenIndependence({ ...base, delayMs: 0 })).toThrow(
+ "positive injected delay"
+ );
+ });
+});
diff --git a/apps/web/src/lib/performance/timeToAnswerPollPhase.test.ts b/apps/web/src/lib/performance/timeToAnswerPollPhase.test.ts
new file mode 100644
index 00000000..bb73e89a
--- /dev/null
+++ b/apps/web/src/lib/performance/timeToAnswerPollPhase.test.ts
@@ -0,0 +1,303 @@
+/**
+ * RED-checks for the stage-5 deterministic poll-phase design (#335, methodology
+ * revision 2).
+ *
+ * The design replaced a uniform random trigger delay with a declared phase grid,
+ * because baseline 3's raw samples proved the jitter did not move the phase: the
+ * reference-25 samples occupied a 120 ms band of a 2000 ms interval. These tests
+ * pin the design (grid shape, assignment, bucketing, normalisation) and, above
+ * all, pin the guards: a sweep confined to a narrow band, a missing phase, an
+ * uncontrolled publication, an observation that did not travel the real API poller
+ * path, or a sample that observed no new revision must all be REJECTED.
+ */
+import { describe, expect, it } from "vitest";
+import {
+ POLL_PHASE_CONTROL_TOLERANCE_MS,
+ POLL_PHASE_DIVISIONS,
+ POLL_PHASE_MIN_SAMPLES_PER_PHASE,
+ assertFreeRunningPhaseSamples,
+ assertControlledPhaseEvidence,
+ assertPollPhaseSweep,
+ declaredPhaseForSample,
+ intendedLatencyMs,
+ observedPhaseBucketMs,
+ phaseMediansMs,
+ phaseNormalizedP95Ms,
+ pollPhaseGridMs,
+ type PollPhaseSweep
+} from "./timeToAnswerPollPhase";
+import { TIME_TO_ANSWER_CONTROLLED_RUNS, TIME_TO_ANSWER_WARMED_SAMPLES } from "./timeToAnswerEvidence";
+
+const INTERVAL = 2000;
+
+/**
+ * A sweep that satisfies the declared design: every phase present, each run
+ * sweeping the grid ascending, observed latency equal to the intended value with a
+ * small, deterministic error so the samples are not identical.
+ */
+function goodSweep(intervalMs = INTERVAL, errorMs = 4): PollPhaseSweep[] {
+ const samples: PollPhaseSweep[] = [];
+ for (let run = 0; run < TIME_TO_ANSWER_CONTROLLED_RUNS; run += 1) {
+ for (let index = 0; index < POLL_PHASE_DIVISIONS; index += 1) {
+ const declaredPhaseMs = declaredPhaseForSample(run, index, intervalMs);
+ const intended = intendedLatencyMs(declaredPhaseMs, intervalMs);
+ const observed = intended + errorMs + run;
+ samples.push({
+ runIndex: run,
+ sampleIndex: index,
+ declaredPhaseMs,
+ intendedPhaseMs: declaredPhaseMs,
+ observedPhaseMs: declaredPhaseMs,
+ intendedLatencyMs: intended,
+ connectedAtMs: 0,
+ predictedPublicationAtMs: 1000,
+ observedPublicationAtMs: 1000,
+ observedObservationAtMs: 1000 + observed,
+ observedLatencyMs: observed,
+ publicationLatencyMs: observed,
+ observedPhaseBucketMs: observedPhaseBucketMs(observed, intervalMs),
+ phaseErrorMs: observed - intended,
+ observedVia: "api-sse",
+ observedRevision: `rev-${run}-${index}`,
+ previousRevision: `rev-${run}-${index}-prev`,
+ phaseControlled: true
+ });
+ }
+ }
+ return samples;
+}
+
+describe("stage-5 declared phase grid", () => {
+ it("INVALIDATES a controlled cell when the requested publication phase was not established", () => {
+ expect(() => assertControlledPhaseEvidence({
+ intendedPhaseMs: 500,
+ observedPhaseMs: 900,
+ publicationLatencyMs: 1500,
+ phaseErrorMs: 0
+ })).toThrow(/could not be established; the cell is invalidated/);
+ });
+
+ it("records the complete controlled-phase audit evidence", () => {
+ const sample = goodSweep()[0]!;
+ expect(sample).toMatchObject({
+ intendedPhaseMs: sample.declaredPhaseMs,
+ observedPhaseMs: sample.declaredPhaseMs,
+ publicationLatencyMs: sample.observedLatencyMs,
+ phaseErrorMs: sample.observedLatencyMs - sample.intendedLatencyMs
+ });
+ });
+
+ it("divides the poll interval into the declared number of phases", () => {
+ const grid = pollPhaseGridMs(INTERVAL);
+ expect(grid).toHaveLength(POLL_PHASE_DIVISIONS);
+ expect([...grid].sort((left, right) => left - right)).toEqual(grid);
+ expect(new Set(grid).size).toBe(grid.length);
+ // Every phase sits strictly inside the interval: a publication landing exactly
+ // on a tick is inherently ambiguous and is deliberately not declared.
+ expect(grid[0]!).toBeGreaterThan(0);
+ expect(grid[grid.length - 1]!).toBeLessThan(INTERVAL);
+ // The declared phases cover essentially the whole interval.
+ const span = grid[grid.length - 1]! - grid[0]!;
+ expect(span / INTERVAL).toBeGreaterThan(0.8);
+ });
+
+ it("maps each declared phase to a strictly decreasing intended latency", () => {
+ const latencies = pollPhaseGridMs(INTERVAL).map((phase) => intendedLatencyMs(phase, INTERVAL));
+ for (let index = 1; index < latencies.length; index += 1) {
+ expect(latencies[index]!).toBeLessThan(latencies[index - 1]!);
+ }
+ expect(latencies[0]!).toBeLessThan(INTERVAL);
+ expect(latencies[latencies.length - 1]!).toBeGreaterThan(0);
+ });
+
+ it("walks the declared grid once per run and repeats it, identically across runs", () => {
+ const grid = pollPhaseGridMs(INTERVAL);
+ for (let index = 0; index < TIME_TO_ANSWER_WARMED_SAMPLES; index += 1) {
+ const runZero = declaredPhaseForSample(0, index, INTERVAL);
+ expect(declaredPhaseForSample(1, index, INTERVAL)).toBe(runZero);
+ expect(declaredPhaseForSample(2, index, INTERVAL)).toBe(runZero);
+ // Fifteen samples against ten divisions: each run covers the whole grid and then
+ // repeats its opening phases, so no phase is left with a single sample.
+ expect(runZero).toBe(grid[index % POLL_PHASE_DIVISIONS]);
+ }
+ const coverage = new Map();
+ for (let run = 0; run < 3; run += 1) {
+ for (let index = 0; index < TIME_TO_ANSWER_WARMED_SAMPLES; index += 1) {
+ const phase = declaredPhaseForSample(run, index, INTERVAL);
+ coverage.set(phase, (coverage.get(phase) ?? 0) + 1);
+ }
+ }
+ expect(coverage.size).toBe(POLL_PHASE_DIVISIONS);
+ expect(Math.min(...coverage.values())).toBeGreaterThanOrEqual(
+ POLL_PHASE_MIN_SAMPLES_PER_PHASE
+ );
+ expect(() => declaredPhaseForSample(0, -1, INTERVAL)).toThrow();
+ expect(() => declaredPhaseForSample(0, 1.5, INTERVAL)).toThrow();
+ });
+
+ it("buckets an observed latency to the nearest declared latency", () => {
+ const grid = pollPhaseGridMs(INTERVAL).map((phase) => intendedLatencyMs(phase, INTERVAL));
+ for (const candidate of grid) {
+ expect(observedPhaseBucketMs(candidate + 3, INTERVAL)).toBe(candidate);
+ expect(observedPhaseBucketMs(candidate - 3, INTERVAL)).toBe(candidate);
+ }
+ });
+});
+
+describe("stage-5 phase sweep validity", () => {
+ it("accepts a sweep that covers the declared grid with a controlled phase", () => {
+ const verdict = assertPollPhaseSweep(goodSweep(), INTERVAL);
+ expect(verdict.divisions).toBe(POLL_PHASE_DIVISIONS);
+ expect(verdict.samples).toBe(POLL_PHASE_DIVISIONS * TIME_TO_ANSWER_CONTROLLED_RUNS);
+ expect(verdict.samplesPerPhase.every((count) => count >= POLL_PHASE_MIN_SAMPLES_PER_PHASE)).toBe(true);
+ expect(verdict.spanShare).toBeGreaterThan(0.5);
+ expect(verdict.directionShare).toBeGreaterThan(0.5);
+ expect(verdict.worstPhaseErrorMs).toBeLessThanOrEqual(POLL_PHASE_CONTROL_TOLERANCE_MS);
+ });
+
+ it("REJECTS a narrow-band sweep — the defect that invalidated baseline 3", () => {
+ // Every sample clustered near one latency: the sample set a random-jitter
+ // harness produced for reference-25 (120 ms of a 2000 ms interval). Which guard
+ // fires first depends on the shape — the declared design is contradicted either
+ // because the phase was not driven or because the spread is too small — but it
+ // is always REJECTED.
+ const narrow = goodSweep().map((sample) => ({ ...sample, observedLatencyMs: 800 + (sample.sampleIndex % 5) }));
+ expect(() => assertPollPhaseSweep(narrow, INTERVAL)).toThrow(
+ /narrow band|phase response|not controlled/
+ );
+ });
+
+ it("REJECTS a sweep that does not show a phase response", () => {
+ const flat = goodSweep(INTERVAL, 0).map((sample) => ({ ...sample, observedLatencyMs: 900 }));
+ expect(() => assertPollPhaseSweep(flat, INTERVAL)).toThrow(
+ /narrow band|phase response|not controlled/
+ );
+ });
+
+ it("REJECTS a missing or thin declared phase", () => {
+ // One run is not a sweep: every declared phase would have a single sample.
+ const singleRun = goodSweep().filter((sample) => sample.runIndex === 0);
+ expect(() => assertPollPhaseSweep(singleRun, INTERVAL)).toThrow(/declared phase|thin/);
+ // A run that lost a phase fails structurally rather than silently reindexing.
+ const gap = goodSweep().filter((sample) => !(sample.runIndex === 0 && sample.sampleIndex === 3));
+ expect(() => assertPollPhaseSweep(gap, INTERVAL)).toThrow(/declared phase/);
+ });
+
+ it("REJECTS a publication phase that was not controlled", () => {
+ const uncontrolled = goodSweep().map((sample) =>
+ sample.sampleIndex === 7
+ ? { ...sample, observedLatencyMs: sample.intendedLatencyMs + 300, phaseErrorMs: 300 }
+ : sample
+ );
+ expect(() => assertPollPhaseSweep(uncontrolled, INTERVAL)).toThrow(/not controlled/);
+ });
+
+ it("REJECTS an observation that did not travel the real API poller path", () => {
+ const shortcut = goodSweep().map((sample) =>
+ sample.sampleIndex === 2
+ ? { ...sample, observedVia: "document-title-poll" }
+ : sample
+ );
+ expect(() => assertPollPhaseSweep(shortcut, INTERVAL)).toThrow(/real API poller path/);
+ });
+
+ it("REJECTS a sample that observed no new revision", () => {
+ const idle = goodSweep().map((sample) =>
+ sample.sampleIndex === 5 ? { ...sample, observedRevision: sample.previousRevision } : sample
+ );
+ expect(() => assertPollPhaseSweep(idle, INTERVAL)).toThrow(/not a publication observation/);
+ });
+
+ it("REJECTS a publication that does not correspond to the controlled trigger", () => {
+ const unrelated = goodSweep().map((sample) =>
+ sample.sampleIndex === 1
+ ? { ...sample, observedPublicationAtMs: sample.predictedPublicationAtMs - INTERVAL }
+ : sample
+ );
+ expect(() => assertPollPhaseSweep(unrelated, INTERVAL)).toThrow(/controlled trigger/);
+ });
+
+ it("REJECTS a declared phase that is not on the declared grid", () => {
+ const offGrid = goodSweep().map((sample) =>
+ sample.sampleIndex === 4 ? { ...sample, declaredPhaseMs: 12.5 } : sample
+ );
+ expect(() => assertPollPhaseSweep(offGrid, INTERVAL)).toThrow(/declared grid|design requires/);
+ });
+});
+
+describe("stage-5 free-running cells", () => {
+ const freeRunning = (count: number, latencyMs: (index: number) => number): PollPhaseSweep[] =>
+ Array.from({ length: count }, (_, index) => ({
+ ...goodSweep()[0]!,
+ runIndex: 0,
+ sampleIndex: index % POLL_PHASE_DIVISIONS,
+ declaredPhaseMs: 0,
+ intendedLatencyMs: latencyMs(index),
+ observedLatencyMs: latencyMs(index),
+ phaseErrorMs: 0,
+ observedRevision: `free-${index}`,
+ previousRevision: `free-${index}-prev`,
+ phaseControlled: false
+ }));
+
+ it("accepts real poller observations with the achieved phase recorded", () => {
+ const verdict = assertFreeRunningPhaseSamples(freeRunning(30, (index) => 100 + index * 50), 30);
+ expect(verdict.samples).toBe(30);
+ expect(verdict.spanMs).toBeGreaterThan(0);
+ expect(verdict.observedPhaseBucketsMs).toHaveLength(30);
+ });
+
+ it("REJECTS a free-running cell with too few samples", () => {
+ expect(() => assertFreeRunningPhaseSamples(freeRunning(5, () => 500), 10)).toThrow(/at least 10 samples/);
+ });
+
+ it("REJECTS a free-running sample that saw no new revision or a non-poller path", () => {
+ const idle = freeRunning(30, () => 500).map((sample, index) =>
+ index === 3 ? { ...sample, observedRevision: sample.previousRevision } : sample
+ );
+ expect(() => assertFreeRunningPhaseSamples(idle, 10)).toThrow(/no new revision/);
+ const shortcut = freeRunning(30, () => 500).map((sample, index) =>
+ index === 7 ? { ...sample, observedVia: "daemon-direct" } : sample
+ );
+ expect(() => assertFreeRunningPhaseSamples(shortcut, 10)).toThrow(/real API poller path/);
+ });
+
+ it("REJECTS a declared sweep over free-running samples", () => {
+ // The two guards must not be interchangeable: a cell whose phase the harness did
+ // not drive cannot claim the sweep's coverage.
+ expect(() => assertPollPhaseSweep(freeRunning(30, () => 500), INTERVAL)).toThrow(/did not drive/);
+ });
+});
+
+describe("stage-5 phase-normalized summary", () => {
+ it("reports a median per declared phase and normalises over the uniform grid", () => {
+ const grid = pollPhaseGridMs(INTERVAL);
+ const runs = [10, 20, 30].map((offset) =>
+ grid.map((phase) => intendedLatencyMs(phase, INTERVAL) + offset)
+ );
+ const medians = phaseMediansMs(runs, INTERVAL);
+ expect(medians).toHaveLength(POLL_PHASE_DIVISIONS);
+ // Each phase's median is its 20 ms sample, and the phases are ordered by
+ // intended latency: the earliest declared phase is the SLOWEST.
+ expect(medians[0]).toBeGreaterThan(medians[medians.length - 1]!);
+ const normalized = phaseNormalizedP95Ms(runs, INTERVAL);
+ // Nearest-rank p95 over 15 phase medians is the largest phase median, so the
+ // normalized figure equals the slowest declared phase's median.
+ expect(normalized).toBe(Math.max(...medians));
+ expect(normalized).toBe(medians[0]);
+ });
+
+ it("uses every repeated positional sample for its declared phase", () => {
+ // Samples 10–14 repeat phases 0–4. Their deliberately large values make a
+ // first-occurrence-only implementation observably wrong.
+ const runs = Array.from({ length: TIME_TO_ANSWER_CONTROLLED_RUNS }, (_, run) =>
+ Array.from({ length: TIME_TO_ANSWER_WARMED_SAMPLES }, (_, sample) =>
+ sample < POLL_PHASE_DIVISIONS ? sample * 10 + run : 1_000 + (sample - POLL_PHASE_DIVISIONS) * 10 + run
+ )
+ );
+ const medians = phaseMediansMs(runs, INTERVAL);
+ expect(medians.slice(0, 5)).toEqual([501, 511, 521, 531, 541]);
+ expect(medians.slice(5)).toEqual([51, 61, 71, 81, 91]);
+ expect(phaseNormalizedP95Ms(runs, INTERVAL)).toBe(541);
+ });
+});
diff --git a/apps/web/src/lib/performance/timeToAnswerPollPhase.ts b/apps/web/src/lib/performance/timeToAnswerPollPhase.ts
new file mode 100644
index 00000000..5ac2c4a1
--- /dev/null
+++ b/apps/web/src/lib/performance/timeToAnswerPollPhase.ts
@@ -0,0 +1,478 @@
+/**
+ * Stage-5 poll-phase design (#335, methodology revision 2).
+ *
+ * Stage 5 measures: authoritative daemon publication -> the Node/SSE layer
+ * observes that revision, through the CURRENT production polling mechanism
+ * (a fixed `DOCKERMAP_SSE_INTERVAL_MS` poller; 2000 ms today).
+ *
+ * The mechanism's latency is a sawtooth in the phase of the publication within a
+ * poll interval, so a capture that lets the phase fall where it likes cannot
+ * characterise it: baseline 3's reference-25 samples occupied a 120 ms band
+ * (6.0 % of the interval) even though the harness slept a uniform sub-interval
+ * delay before each trigger, because both the daemon's refresh loop and the
+ * API's poller are fixed 2 s loops and the measured gap was their phase offset.
+ *
+ * This module therefore declares an EXPLICIT deterministic phase sweep instead
+ * of an assumed uniform random distribution:
+ *
+ * - the poll interval is divided into `POLL_PHASE_DIVISIONS` equal parts, giving
+ * one declared phase per recorded sample per run (15 phases, 15 samples);
+ * - phase `p` places the publication at `(p + 0.5) * interval / divisions`, so the
+ * intended latency is `interval - that offset`: the sweep covers the interval
+ * from half a division to `divisions - 0.5` divisions, and no phase sits on a
+ * poll tick boundary, where a publication would be inherently ambiguous;
+ * - each controlled run sweeps the phases in ascending order, so each phase has
+ * exactly `TIME_TO_ANSWER_CONTROLLED_RUNS` samples per cell and the phase of
+ * every raw sample is recoverable from its position in the run;
+ * - the phase is DRIVEN, not hoped for: the harness connects its observation
+ * stream at a computed instant so that the next poll tick after the predicted
+ * publication falls at the intended latency, then verifies the result.
+ *
+ * The summary derived from this design is a PHASE-NORMALIZED figure: it weights
+ * the declared phases uniformly to characterise the latency imposed by the fixed
+ * polling mechanism. It is not an observed user-traffic distribution and not
+ * network latency.
+ */
+
+/** Equal parts the poll interval is divided into; one declared phase per sample per run. */
+/**
+ * Declared divisions of the poll interval. Ten gives a 200 ms grid step, which is
+ * what the mechanism allows: the observation tick is a Node timer that drifts under
+ * load (measured +52 ms at 250 containers), so the step must exceed the achievable
+ * control precision by a margin or neighbouring phases blur together. Ten divisions
+ * still sweep 90% of the interval (latencies 100–1900 ms).
+ */
+export const POLL_PHASE_DIVISIONS = 10;
+
+/**
+ * How far an observed sample may sit from its declared phase before the cell is
+ * invalid. The grid spacing is `interval / divisions` (133.3 ms at 2000 ms), so
+ * the tolerance keeps adjacent phases distinguishable while absorbing the few
+ * milliseconds of publication-grid drift and detection delay.
+ */
+/**
+ * Declared control tolerance: how far the observed latency may sit from the intended
+ * one before a controlled sample is rejected as uncontrolled. Half the grid step
+ * (100 ms) is the mathematical limit — anything larger could bucket into a
+ * neighbouring phase — and 90 ms leaves room for the poll timer's real drift.
+ */
+export const POLL_PHASE_CONTROL_TOLERANCE_MS = 90;
+
+/**
+ * The observed sweep must span at least this share of the poll interval, and the
+ * smallest publication offset must be at least this much slower than the largest
+ * one. A sweep confined to a narrow band cannot satisfy either, which is exactly
+ * the defect that invalidated baseline 3's reference cells.
+ */
+export const POLL_PHASE_MIN_SPAN_SHARE = 0.5;
+export const POLL_PHASE_MIN_DIRECTION_SHARE = 0.5;
+
+/** Every declared phase must appear at least this many times in a controlled cell. */
+export const POLL_PHASE_MIN_SAMPLES_PER_PHASE = 2;
+
+/**
+ * Cells whose publication the harness can actually trigger, and which therefore get
+ * the declared phase sweep.
+ *
+ * The two provider-state fixtures are NOT here on purpose. Their revisions advance
+ * from the daemon's own host provider collection — the harness has no input that
+ * makes the daemon publish at a chosen instant — so their stage-5 samples are
+ * FREE-RUNNING: the phase each sample achieved is recorded, the sweep's coverage and
+ * direction guards do not apply, and no phase-normalized figure is derived for them.
+ * They still measure today's real poll wait, and they are still declared in the
+ * matrix; what they cannot do is place the publication.
+ */
+export const POLL_PHASE_CONTROLLED_FIXTURES = [
+ "reference-25",
+ "reference-100",
+ "reference-250",
+ "docker-topology-change"
+] as const;
+
+export function isPhaseControlledFixture(fixture: string): boolean {
+ return (POLL_PHASE_CONTROLLED_FIXTURES as readonly string[]).includes(fixture);
+}
+
+export interface PollPhaseSweep {
+ runIndex: number;
+ sampleIndex: number;
+ /** Declared publication offset after the enclosing poll tick. */
+ declaredPhaseMs: number;
+ /** Explicit audit name for the phase the harness requested. */
+ intendedPhaseMs: number;
+ /** Publication phase actually witnessed from the connected observation stream. */
+ observedPhaseMs: number;
+ /** Latency the declared phase should produce. */
+ intendedLatencyMs: number;
+ /** When the harness connected its observation stream, relative to the run clock. */
+ connectedAtMs: number;
+ /** Predicted publication instant for this sample, relative to the run clock. */
+ predictedPublicationAtMs: number;
+ /** Observed publication instant, relative to the run clock. */
+ observedPublicationAtMs: number;
+ /** Observed poll tick that carried the revision, relative to the run clock. */
+ observedObservationAtMs: number;
+ observedLatencyMs: number;
+ /** Publication-to-observation latency on the real API poller path. */
+ publicationLatencyMs: number;
+ /** The declared phase whose intended latency is nearest to the observed latency. */
+ observedPhaseBucketMs: number;
+ /** Observed minus intended latency. */
+ phaseErrorMs: number;
+ /** How the observation reached the harness. Only the real poller path is valid. */
+ observedVia: string;
+ /** The revision observed, and the revision that was current before the sample. */
+ observedRevision: string;
+ previousRevision: string;
+ /**
+ * Whether the harness drove this sample's publication phase. Free-running cells
+ * (the provider-state fixtures) record the phase they achieved instead, and are
+ * excluded from the sweep's coverage and direction guards.
+ */
+ phaseControlled: boolean;
+}
+
+/**
+ * Fail closed before a controlled sample is recorded. A phase-controlled cell
+ * may only claim control when the witnessed publication phase and poll latency
+ * both match the requested phase within the declared tolerance.
+ */
+export function assertControlledPhaseEvidence(evidence: Pick): void {
+ const values = {
+ intendedPhaseMs: evidence.intendedPhaseMs,
+ observedPhaseMs: evidence.observedPhaseMs,
+ publicationLatencyMs: evidence.publicationLatencyMs,
+ phaseErrorMs: evidence.phaseErrorMs
+ };
+ for (const [name, value] of Object.entries(values)) {
+ if (!Number.isFinite(value)) throw new Error(`stage-5 ${name} is not finite`);
+ }
+ if (Math.abs(evidence.observedPhaseMs - evidence.intendedPhaseMs) > POLL_PHASE_CONTROL_TOLERANCE_MS) {
+ throw new Error("stage-5 requested publication phase could not be established; the cell is invalidated");
+ }
+ if (Math.abs(evidence.phaseErrorMs) > POLL_PHASE_CONTROL_TOLERANCE_MS) {
+ throw new Error("stage-5 requested poll phase could not be established; the cell is invalidated");
+ }
+}
+
+export interface PollPhaseValidity {
+ divisions: number;
+ intervalMs: number;
+ samples: number;
+ declaredPhasesMs: readonly number[];
+ observedPhasesMs: readonly number[];
+ minObservedLatencyMs: number;
+ maxObservedLatencyMs: number;
+ spanMs: number;
+ spanShare: number;
+ slowestPhaseMedianMs: number;
+ fastestPhaseMedianMs: number;
+ directionShare: number;
+ worstPhaseErrorMs: number;
+ samplesPerPhase: readonly number[];
+}
+
+function assertInterval(intervalMs: number): void {
+ if (!Number.isFinite(intervalMs) || intervalMs <= 0) {
+ throw new Error("the poll interval must be a finite positive number of milliseconds");
+ }
+}
+
+/**
+ * Declared publication offsets within the poll interval, ascending.
+ *
+ * Phase `p` places the publication at `(p + 0.5) * interval / divisions`, so every
+ * declared phase sits at the CENTRE of its division: never on a poll tick boundary
+ * (where a publication is inherently ambiguous) and never at the very end of the
+ * interval.
+ */
+export function pollPhaseGridMs(intervalMs: number): number[] {
+ assertInterval(intervalMs);
+ const step = intervalMs / POLL_PHASE_DIVISIONS;
+ return Array.from({ length: POLL_PHASE_DIVISIONS }, (_, index) => (index + 0.5) * step);
+}
+
+/** Declared latency for a publication offset: the wait until the next poll tick. */
+export function intendedLatencyMs(phaseMs: number, intervalMs: number): number {
+ assertInterval(intervalMs);
+ if (!Number.isFinite(phaseMs) || phaseMs <= 0 || phaseMs >= intervalMs) {
+ throw new Error("a declared phase must sit strictly inside the poll interval");
+ }
+ return intervalMs - phaseMs;
+}
+
+/** The declared phase for a sample: run `r` sweeps the grid ascending from index 0. */
+export function declaredPhaseIndexForSample(_runIndex: number, sampleIndex: number): number {
+ if (!Number.isInteger(sampleIndex) || sampleIndex < 0) {
+ throw new Error("a stage-5 sample index must be a non-negative integer");
+ }
+ // Fifteen recorded samples against ten declared divisions: each run walks the whole
+ // grid and then repeats its first five phases, so across the three controlled runs
+ // every declared phase carries at least three samples.
+ return sampleIndex % POLL_PHASE_DIVISIONS;
+}
+
+/** The declared phase a raw sample must have produced, from its position in its run. */
+export function declaredPhaseForSample(runIndex: number, sampleIndex: number, intervalMs: number): number {
+ return pollPhaseGridMs(intervalMs)[declaredPhaseIndexForSample(runIndex, sampleIndex)]!;
+}
+
+/** Nearest declared latency for an observed latency: the observed phase bucket. */
+export function observedPhaseBucketMs(observedLatencyMs: number, intervalMs: number): number {
+ const grid = pollPhaseGridMs(intervalMs).map((phase) => intendedLatencyMs(phase, intervalMs));
+ let best = grid[0]!;
+ let bestDistance = Number.POSITIVE_INFINITY;
+ for (const candidate of grid) {
+ const distance = Math.abs(candidate - observedLatencyMs);
+ if (distance < bestDistance) {
+ best = candidate;
+ bestDistance = distance;
+ }
+ }
+ return best;
+}
+
+function median(values: readonly number[]): number {
+ const ordered = [...values].sort((left, right) => left - right);
+ const middle = Math.floor(ordered.length / 2);
+ return ordered.length % 2 === 1 ? ordered[middle]! : (ordered[middle - 1]! + ordered[middle]!) / 2;
+}
+
+/**
+ * Nearest-rank p95 over the declared-phase medians: the phase-normalized figure.
+ * Each declared phase is weighted equally, which is a statement about the
+ * MECHANISM under uniformly sampled phase offsets — never about observed
+ * user-traffic distribution, and never about network distance.
+ */
+export function phaseNormalizedP95Ms(runs: readonly (readonly number[])[], intervalMs: number): number {
+ const perPhaseMedians = phaseMediansMs(runs, intervalMs);
+ const ordered = [...perPhaseMedians].sort((left, right) => left - right);
+ return ordered[Math.ceil(ordered.length * 0.95) - 1]!;
+}
+
+/** Median observed latency per declared phase, ascending by phase. */
+export function phaseMediansMs(runs: readonly (readonly number[])[], intervalMs: number): number[] {
+ const grid = pollPhaseGridMs(intervalMs);
+ return grid.map((_phase, phaseIndex) => {
+ const samples = runs.flatMap((run, runIndex) =>
+ run.filter(
+ (value, sampleIndex): value is number =>
+ declaredPhaseIndexForSample(runIndex, sampleIndex) === phaseIndex &&
+ typeof value === "number" &&
+ Number.isFinite(value)
+ )
+ );
+ if (samples.length === 0) {
+ throw new Error(`no samples exist for declared phase ${phaseIndex + 1}/${POLL_PHASE_DIVISIONS}`);
+ }
+ return median(samples);
+ });
+}
+
+/**
+ * Enforce the declared experimental design on one stage-5 cell. Throws when the
+ * sweep does not cover the interval, when a declared phase is missing, when the
+ * phase was not actually driven to its declared offset, when an observation did
+ * not arrive through the real API poller path, or when the observed spread is
+ * too narrow to characterise a sawtooth whose range is one poll interval.
+ */
+export function assertPollPhaseSweep(
+ samples: readonly PollPhaseSweep[],
+ intervalMs: number,
+ minSamplesPerPhase = POLL_PHASE_MIN_SAMPLES_PER_PHASE
+): PollPhaseValidity {
+ assertInterval(intervalMs);
+ if (samples.length === 0) throw new Error("the stage-5 phase sweep has no samples");
+ const grid = pollPhaseGridMs(intervalMs);
+
+ const declaredPhasesMs: number[] = [];
+ const observedPhasesMs: number[] = [];
+ const samplesPerPhase = grid.map(() => 0);
+ let worstPhaseErrorMs = 0;
+ let minObservedLatencyMs = Number.POSITIVE_INFINITY;
+ let maxObservedLatencyMs = Number.NEGATIVE_INFINITY;
+
+ for (const sample of samples) {
+ if (!Number.isFinite(sample.observedLatencyMs) || sample.observedLatencyMs < 0) {
+ throw new Error("a stage-5 sample has a non-finite observed latency");
+ }
+ if (!sample.phaseControlled) {
+ throw new Error(
+ "the declared phase sweep was applied to a sample the harness did not drive; " +
+ "free-running cells use the free-running guard instead"
+ );
+ }
+ if (sample.observedVia !== "api-sse") {
+ throw new Error(
+ `a stage-5 sample was observed via ${sample.observedVia} instead of the real API poller path`
+ );
+ }
+ if (!sample.observedRevision || sample.observedRevision === sample.previousRevision) {
+ throw new Error("a stage-5 sample did not observe a new revision, so it is not a publication observation");
+ }
+ if (sample.observedPublicationAtMs < sample.predictedPublicationAtMs - intervalMs / 2) {
+ throw new Error("a stage-5 sample's observed publication does not correspond to the controlled trigger");
+ }
+ const expectedPhase = declaredPhaseForSample(sample.runIndex, sample.sampleIndex, intervalMs);
+ if (Math.abs(expectedPhase - sample.declaredPhaseMs) > 1e-6) {
+ throw new Error(
+ `stage-5 sample ${sample.sampleIndex} of run ${sample.runIndex} declares phase ${sample.declaredPhaseMs} ` +
+ `but the design requires ${expectedPhase}`
+ );
+ }
+ const phaseErrorMs = sample.observedLatencyMs - sample.intendedLatencyMs;
+ worstPhaseErrorMs = Math.max(worstPhaseErrorMs, Math.abs(phaseErrorMs));
+ if (Math.abs(phaseErrorMs) > POLL_PHASE_CONTROL_TOLERANCE_MS) {
+ throw new Error(
+ `stage-5 phase ${sample.declaredPhaseMs.toFixed(1)} ms produced ${sample.observedLatencyMs.toFixed(1)} ms ` +
+ `instead of ${sample.intendedLatencyMs.toFixed(1)} ms (error ${phaseErrorMs.toFixed(1)} ms exceeds the ` +
+ `${POLL_PHASE_CONTROL_TOLERANCE_MS} ms tolerance): the publication phase was not controlled`
+ );
+ }
+ const bucket = observedPhaseBucketMs(sample.observedLatencyMs, intervalMs);
+ const declaredPhaseIndex = declaredPhaseIndexOfValue(sample.declaredPhaseMs, grid);
+ if (declaredPhaseIndex < 0) {
+ throw new Error(`stage-5 declared phase ${sample.declaredPhaseMs} is not on the declared grid`);
+ }
+ declaredPhasesMs.push(sample.declaredPhaseMs);
+ observedPhasesMs.push(bucket);
+ samplesPerPhase[declaredPhaseIndex] = (samplesPerPhase[declaredPhaseIndex] ?? 0) + 1;
+ minObservedLatencyMs = Math.min(minObservedLatencyMs, sample.observedLatencyMs);
+ maxObservedLatencyMs = Math.max(maxObservedLatencyMs, sample.observedLatencyMs);
+ }
+
+ const sparse = samplesPerPhase
+ .map((count, index) => ({ count, index }))
+ .filter((entry) => entry.count < minSamplesPerPhase);
+ if (sparse.length > 0) {
+ throw new Error(
+ `the stage-5 sweep must cover every declared phase at least ${minSamplesPerPhase} times; ` +
+ `missing or thin phases: ${sparse.map((entry) => entry.index + 1).join(", ")}`
+ );
+ }
+
+ const spanMs = maxObservedLatencyMs - minObservedLatencyMs;
+ const spanShare = spanMs / intervalMs;
+ if (spanShare < POLL_PHASE_MIN_SPAN_SHARE) {
+ throw new Error(
+ `the stage-5 sweep spans only ${spanMs.toFixed(1)} ms (${(spanShare * 100).toFixed(1)} % of the ` +
+ `${intervalMs} ms interval): a narrow band cannot characterise the polling mechanism's phase response`
+ );
+ }
+
+ const perPhase = phaseMediansMs(
+ groupRunsByPhase(samples, intervalMs),
+ intervalMs
+ );
+ const slowestPhaseMedianMs = perPhase[0]!;
+ const fastestPhaseMedianMs = perPhase[perPhase.length - 1]!;
+ const directionShare = (slowestPhaseMedianMs - fastestPhaseMedianMs) / intervalMs;
+ if (directionShare < POLL_PHASE_MIN_DIRECTION_SHARE) {
+ throw new Error(
+ `the stage-5 sweep does not show a phase response: the earliest declared phase median ` +
+ `(${slowestPhaseMedianMs.toFixed(1)} ms) is only ${(directionShare * 100).toFixed(1)} % of an interval ` +
+ `above the latest (${fastestPhaseMedianMs.toFixed(1)} ms)`
+ );
+ }
+
+ return {
+ divisions: POLL_PHASE_DIVISIONS,
+ intervalMs,
+ samples: samples.length,
+ declaredPhasesMs,
+ observedPhasesMs,
+ minObservedLatencyMs,
+ maxObservedLatencyMs,
+ spanMs,
+ spanShare,
+ slowestPhaseMedianMs,
+ fastestPhaseMedianMs,
+ directionShare,
+ worstPhaseErrorMs,
+ samplesPerPhase
+ };
+}
+
+function declaredPhaseIndexOfValue(phaseMs: number, grid: readonly number[]): number {
+ return grid.findIndex((candidate) => Math.abs(candidate - phaseMs) < 1e-6);
+}
+
+export interface FreeRunningPhaseValidity {
+ samples: number;
+ minObservedLatencyMs: number;
+ maxObservedLatencyMs: number;
+ spanMs: number;
+ observedPhaseBucketsMs: readonly number[];
+}
+
+/**
+ * Guard for FREE-RUNNING stage-5 cells — the provider-state fixtures, whose
+ * publications the harness cannot place. Coverage and direction are NOT required
+ * (the design does not claim to control the phase there), but the sample must still
+ * be a real observation: a new revision, seen through the real API poller path, with
+ * the phase it achieved recorded. A cell that recorded no usable samples, or that
+ * saw a non-poller observation, is rejected.
+ */
+export function assertFreeRunningPhaseSamples(
+ samples: readonly PollPhaseSweep[],
+ minimumSamples = 10
+): FreeRunningPhaseValidity {
+ if (samples.length < minimumSamples) {
+ throw new Error(
+ `a free-running stage-5 cell needs at least ${minimumSamples} samples, got ${samples.length}`
+ );
+ }
+ let minObservedLatencyMs = Number.POSITIVE_INFINITY;
+ let maxObservedLatencyMs = Number.NEGATIVE_INFINITY;
+ const observedPhaseBucketsMs: number[] = [];
+ for (const sample of samples) {
+ if (sample.phaseControlled) {
+ throw new Error("a free-running cell must not contain phase-controlled samples");
+ }
+ if (sample.observedVia !== "api-sse") {
+ throw new Error("a free-running stage-5 sample did not arrive through the real API poller path");
+ }
+ if (!sample.observedRevision || sample.observedRevision === sample.previousRevision) {
+ throw new Error("a free-running stage-5 sample observed no new revision");
+ }
+ if (!Number.isFinite(sample.observedLatencyMs) || sample.observedLatencyMs < 0) {
+ throw new Error("a free-running stage-5 sample has a non-finite observed latency");
+ }
+ observedPhaseBucketsMs.push(sample.observedPhaseBucketMs);
+ minObservedLatencyMs = Math.min(minObservedLatencyMs, sample.observedLatencyMs);
+ maxObservedLatencyMs = Math.max(maxObservedLatencyMs, sample.observedLatencyMs);
+ }
+ return {
+ samples: samples.length,
+ minObservedLatencyMs,
+ maxObservedLatencyMs,
+ spanMs: maxObservedLatencyMs - minObservedLatencyMs,
+ observedPhaseBucketsMs
+ };
+}
+
+/** Rebuild per-run sample arrays from sweep records, so the shared math can be reused. */
+function groupRunsByPhase(samples: readonly PollPhaseSweep[], intervalMs: number): number[][] {
+ const runIndexes = [...new Set(samples.map((sample) => sample.runIndex))].sort((left, right) => left - right);
+ return runIndexes.map((runIndex) => {
+ const run = samples
+ .filter((sample) => sample.runIndex === runIndex)
+ .sort((left, right) => left.sampleIndex - right.sampleIndex);
+ const values: number[] = [];
+ for (const sample of run) {
+ if (sample.sampleIndex !== values.length) {
+ throw new Error(`run ${runIndex} has a missing or duplicate declared phase sample index`);
+ }
+ assertControlledPhaseEvidence(sample);
+ // The sample's own recorded phase decides its slot, so the shared math cannot
+ // silently disagree with the declaration the capture used.
+ if (Math.abs(sample.declaredPhaseMs - declaredPhaseForSample(runIndex, sample.sampleIndex, intervalMs)) >= 1e-6) {
+ throw new Error(`run ${runIndex} has an incorrectly declared stage-5 phase`);
+ }
+ values.push(sample.observedLatencyMs);
+ }
+ return values;
+ });
+}
diff --git a/apps/web/src/lib/performance/timeToAnswerPromotion.test.ts b/apps/web/src/lib/performance/timeToAnswerPromotion.test.ts
new file mode 100644
index 00000000..1474467c
--- /dev/null
+++ b/apps/web/src/lib/performance/timeToAnswerPromotion.test.ts
@@ -0,0 +1,474 @@
+/**
+ * RED-checks for the time-to-answer promotion gate (#335).
+ *
+ * Every case here must fail for an EVIDENCE reason — a bad comparison — and not
+ * because the fixture happens to be malformed in some unrelated way. The
+ * candidate in each rejection case is otherwise a complete, valid artifact.
+ */
+import { describe, expect, it } from "vitest";
+import {
+ TIME_TO_ANSWER_BASELINE,
+ TIME_TO_ANSWER_CONTROLLED_RUNS,
+ TIME_TO_ANSWER_MATRIX,
+ TIME_TO_ANSWER_METHODOLOGY,
+ TIME_TO_ANSWER_WARMED_SAMPLES,
+ TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS,
+ TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS,
+ TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES,
+ TIME_TO_ANSWER_WARM_UP_METRICS,
+ deriveFrozenWarmUpCount,
+ deriveWarmUpCalibrationReport,
+ assertTimeToAnswerPromotion,
+ assertDaemonBinaryProvenance,
+ compatibleTimeToAnswerEnvironment,
+ isScenarioCell,
+ splitWarmedObservations,
+ summarizeTimeToAnswerStage,
+ TIME_TO_ANSWER_STAGE_KIND,
+ TIME_TO_ANSWER_STAGES,
+ timeToAnswerLimit,
+ validateTimeToAnswerEvidence
+} from "./timeToAnswerEvidence";
+
+const environment = {
+ runnerClass: "linux-x86_64-dedicated",
+ cpuClass: "cpus-16vcpu",
+ osImage: "ubuntu-26.04",
+ osKernel: "7.0.0-31-generic",
+ nodeRevision: "22.23.2",
+ rustRevision: "1.88.0",
+ dockerRevision: "29.8.1",
+ ssePollIntervalMs: "2000",
+ daemonBinarySha256: "eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee",
+ daemonBinaryBuild: "cargo-build-release-locked-p-dockermap-daemon",
+ cargoRevision: "cargo-1.88.0",
+ harnessRevision: "dddddddddddddddddddddddddddddddddddddddd",
+ browserEngine: "chromium",
+ browserRevision: "1.61.0",
+ browserFlags: ["--disable-background-networking"],
+ fontEnvironment: "system-default",
+ buildMode: "production",
+ fixtureRevision: "dockermap-v1/time-to-answer-fixtures-1",
+ sourceRevision: "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
+ methodologyVersion: TIME_TO_ANSWER_METHODOLOGY
+};
+
+/** 15 finite non-negative warmed samples with a per-run offset. */
+function samples(base: number): number[] {
+ return Array.from({ length: TIME_TO_ANSWER_WARMED_SAMPLES }, (_, index) => base + index * 0.1);
+}
+
+function artifact(overrides: { environment?: Record; records?: unknown[] } = {}) {
+ const provenance = (record: any) => ({
+ ...record,
+ measurementProtocol: record.measurementProtocol ?? protocol(record.fixture, record.stage),
+ sourceEvidenceFile: record.sourceEvidenceFile ?? evidenceFile(record.fixture, record.stage),
+ checkpointSha: record.checkpointSha ?? "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"
+ });
+ return {
+ baseline: TIME_TO_ANSWER_BASELINE,
+ environment: { ...environment, ...(overrides.environment ?? {}) },
+ records:
+ (overrides.records ? overrides.records.map(provenance) :
+ TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => ({
+fixture,
+stage,
+ measurementProtocol: protocol(fixture, stage),
+ sourceEvidenceFile: evidenceFile(fixture, stage),
+checkpointSha: "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
+runs: [samples(10), samples(11), samples(12)]
+ })))
+ };
+}
+
+function protocol(fixture: string, stage: string): string {
+ void fixture;
+ if (stage === "publicationToNodeObservationMs") return "controlled-poll-phase";
+ return "end-to-end";
+}
+
+function evidenceFile(fixture: string, stage: string): string {
+ return protocol(fixture, stage) === "controlled-poll-phase" ? "stage-five.raw.json" : "general.raw.json";
+}
+
+function candidate(overrides: { environment?: Record; records?: unknown[] } = {}) {
+ return artifact({
+ ...overrides,
+ environment: { sourceRevision: "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", ...(overrides.environment ?? {}) }
+ });
+}
+
+describe("time-to-answer promotion gate", () => {
+ it("uses the declared fixed 60-observation burn-in without a per-metric table", () => {
+ expect(TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS).toBe(60);
+ });
+
+ it("derives the earliest sustained calibration point, maximum fixture count, and fixed margin", () => {
+ const stable = Array.from({ length: TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS }, () => 10);
+ // This fixture is unsettled at candidates 2 and 3, then stationary through
+ // every eligible position. The metric result is its earliest stable point +2.
+ stable[0] = 100;
+ stable[1] = 100;
+ stable[2] = 100;
+ stable[3] = 100;
+ const derived = deriveFrozenWarmUpCount([
+ { fixture: "reference-25", metric: "dockerObservationMs", observations: Array.from({ length: TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS }, () => 10) },
+ { fixture: "reference-100", metric: "dockerObservationMs", observations: stable },
+ { fixture: "reference-250", metric: "dockerObservationMs", observations: Array.from({ length: TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS }, () => 10) }
+ ]);
+ expect(derived.fixtureCounts).toEqual({ "reference-25": 2, "reference-100": 6, "reference-250": 2 });
+ expect(derived.frozenWarmUpCount).toBe(8);
+ });
+
+ it("fails rather than extrapolating when the safety margin is not evidence-backed", () => {
+ const observations = Array.from({ length: TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS }, () => 10);
+ // Only candidate 45 is stationary, so 45 + the fixed margin exceeds the
+ // final eligible position (45) and must not become a frozen count.
+ for (let index = 0; index < 43; index += 1) observations[index] = 100;
+ expect(() => deriveFrozenWarmUpCount(TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES.map((fixture) => ({ fixture, metric: "dockerObservationMs", observations })))).toThrow("calibration conflict");
+ });
+
+ it("reports every metric on conflict and never makes a partial table authoritative", () => {
+ const stable = Array.from({ length: TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS }, () => 10);
+ const conflicted = [...stable];
+ // Stable only at the final eligible candidate: +2 then lacks a following 15.
+ for (let index = 0; index < 43; index += 1) conflicted[index] = 100;
+ const cells = TIME_TO_ANSWER_WARM_UP_METRICS.flatMap((metric) => TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES.map((fixture) => ({ fixture, metric, observations: metric === "dockerObservationMs" && fixture === "reference-25" ? conflicted : stable })));
+ const report = deriveWarmUpCalibrationReport(cells);
+ expect(report.verdict).toBe("CONFLICT");
+ expect(report.metrics).toHaveLength(TIME_TO_ANSWER_WARM_UP_METRICS.length);
+ expect(report.metrics.find((metric) => metric.metric === "dockerObservationMs")).toMatchObject({ verdict: "CONFLICT", proposedWarmUpCount: 47, requiredEvidenceLength: 62, evidenceBacked: false });
+ expect(report.nonAuthoritativeProposedWarmUpCounts).toEqual({});
+ expect(report.metrics.find((metric) => metric.metric === "buildModelMs")?.verdict).toBe("PASS");
+ });
+
+ it("records candidate ratios and atomically proposes every count only on complete success", () => {
+ const observations = Array.from({ length: TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS }, () => 10);
+ const report = deriveWarmUpCalibrationReport(TIME_TO_ANSWER_WARM_UP_METRICS.flatMap((metric) => TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES.map((fixture) => ({ fixture, metric, observations }))));
+ expect(report.verdict).toBe("PASS");
+ expect(Object.keys(report.nonAuthoritativeProposedWarmUpCounts)).toEqual(TIME_TO_ANSWER_WARM_UP_METRICS);
+ expect(report.metrics[0]?.fixtures[0]?.stableWarmUpCount).toBe(2);
+ expect(report.metrics[0]?.fixtures[0]?.candidates[0]).toMatchObject({ candidate: 2, ratio: 1, inBand: true, sustained: true });
+ });
+
+ it("rejects a calibration that omits or adds a reference fixture", () => {
+ const observations = Array.from({ length: TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS }, () => 10);
+ expect(() => deriveFrozenWarmUpCount([
+ { fixture: "reference-25", metric: "dockerObservationMs", observations },
+ { fixture: "reference-100", metric: "dockerObservationMs", observations }
+ ])).toThrow("every declared reference fixture");
+ expect(() => deriveFrozenWarmUpCount([
+ ...TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES.map((fixture) => ({ fixture, metric: "dockerObservationMs", observations })),
+ { fixture: "docker-topology-change", metric: "dockerObservationMs", observations }
+ ])).toThrow("every declared reference fixture");
+ });
+
+ it("accepts a compatible candidate inside the reviewed budget", () => {
+ expect(() => assertTimeToAnswerPromotion(artifact(), candidate())).not.toThrow();
+ });
+
+ it("accepts the provenance differences a candidate may carry: source revision, harness revision, rebuilt digest", () => {
+ // Source and harness revisions differ by construction, and the daemon binary is
+ // REBUILT from the candidate checkout — so a byte-identical digest is not even
+ // reproducible across a changed CARGO_HOME. Requiring any of them to match would
+ // make every candidate that touches the product or the harness uncomparable.
+ const comparable = candidate({
+ environment: {
+ sourceRevision: "cccccccccccccccccccccccccccccccccccccccc",
+ harnessRevision: "ab".repeat(20),
+ daemonBinarySha256: "f".repeat(64)
+ }
+ });
+ expect(() => assertTimeToAnswerPromotion(artifact(), comparable)).not.toThrow();
+ const baseline = validateTimeToAnswerEvidence(artifact());
+ const validated = validateTimeToAnswerEvidence(comparable);
+ expect(compatibleTimeToAnswerEnvironment(baseline.environment, validated.environment)).toBe(true);
+ expect(validated.environment.daemonBinarySha256).toBe("f".repeat(64));
+ });
+
+ it("rejects a candidate measured under a different methodology version", () => {
+ const other = candidate({ environment: { methodologyVersion: "dockermap-v1/time-to-answer-methodology-1" } });
+ expect(
+ compatibleTimeToAnswerEnvironment(
+ validateTimeToAnswerEvidence(artifact()).environment,
+ validateTimeToAnswerEvidence(other).environment
+ )
+ ).toBe(false);
+ expect(() => assertTimeToAnswerPromotion(artifact(), other)).toThrow(
+ "does not match the pinned baseline environment"
+ );
+ });
+
+ it("rejects a candidate above the reviewed budget, naming the cell", () => {
+ const slow = candidate({
+ records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) =>
+ fixture === "reference-250" && stage === "dockerObservationMs"
+ ? { fixture, stage, runs: [samples(1_000), samples(1_000), samples(1_000)] }
+ : { fixture, stage, runs: [samples(10), samples(11), samples(12)] }
+ )
+ });
+ expect(() => assertTimeToAnswerPromotion(artifact(), slow)).toThrow("promotion limit");
+ // The limit itself is the reviewed rule, not an invented constant.
+ expect(timeToAnswerLimit(10)).toBe(12.5);
+ expect(timeToAnswerLimit(1)).toBe(3);
+ });
+
+ it("rejects a slow cell the budget tolerates only just", () => {
+ const limit = timeToAnswerLimit(12);
+ const pass = candidate({
+ records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) =>
+ fixture === "reference-100" && stage === "buildModelMs"
+ ? { fixture, stage, runs: [[limit, limit, limit, ...Array(12).fill(limit)], [limit, limit, limit, ...Array(12).fill(limit)], [limit, limit, limit, ...Array(12).fill(limit)]] }
+ : { fixture, stage, runs: [samples(1), samples(1), samples(1)] }
+ )
+ });
+ expect(() => assertTimeToAnswerPromotion(artifact(), pass)).not.toThrow();
+ const fail = candidate({
+ records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) =>
+ fixture === "reference-100" && stage === "buildModelMs"
+ ? { fixture, stage, runs: [[limit + 1, ...Array(14).fill(limit + 1)], [limit + 1, ...Array(14).fill(limit + 1)], [limit + 1, ...Array(14).fill(limit + 1)]] }
+ : { fixture, stage, runs: [samples(1), samples(1), samples(1)] }
+ )
+ });
+ expect(() => assertTimeToAnswerPromotion(artifact(), fail)).toThrow("promotion limit");
+ });
+
+ it.each([
+ ["runner class", "runnerClass", "some-other-runner"],
+ ["cpu class", "cpuClass", "cpus-2vcpu"],
+ ["os image", "osImage", "debian-13"],
+ ["os kernel", "osKernel", "6.8.0-31-generic"],
+ ["node revision", "nodeRevision", "20.11.0"],
+ ["rust revision", "rustRevision", "1.80.0"],
+ ["chromium revision", "browserRevision", "1.50.0"],
+ ["fixture revision", "fixtureRevision", "dockermap-v1/other-fixtures"],
+ ["sse poll interval", "ssePollIntervalMs", "1000"],
+ ["font environment", "fontEnvironment", "different-fonts"]
+ ])("rejects a candidate whose %s does not match the pinned baseline", (_label, key, value) => {
+ expect(() => assertTimeToAnswerPromotion(artifact(), candidate({ environment: { [key]: value } }))).toThrow(
+ "does not match the pinned baseline environment"
+ );
+ });
+
+ it("treats dockerRevision as informational: a host engine change must not fail a comparison", () => {
+ // No measured stage exercises the host Docker daemon — the capture runs
+ // against the deterministic fixture daemon — so pinning it as a
+ // compatibility key would reject a candidate for an untouched dimension.
+ expect(() =>
+ assertTimeToAnswerPromotion(artifact(), candidate({ environment: { dockerRevision: "30.1.0" } }))
+ ).not.toThrow();
+ });
+
+ it("pins the median-of-three aggregation, not the first run or the pooled mean", () => {
+ // One slow run and two fast runs: the median must pass, while a first-run
+ // p95 or a pooled mean would exceed the budget.
+ const slowFirst = candidate({
+ records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) =>
+ fixture === "reference-250" && stage === "commandQueryMs"
+ ? {
+ fixture,
+ stage,
+ runs: [
+ [...Array(15).fill(500)],
+ [...Array(15).fill(10)],
+ [...Array(15).fill(10)]
+ ]
+ }
+ : { fixture, stage, runs: [samples(10), samples(11), samples(12)] }
+ )
+ });
+
+ expect(() => assertTimeToAnswerPromotion(artifact(), slowFirst)).not.toThrow();
+
+ // Two slow runs and one fast run: the median is slow, so it must fail.
+ const slowMajority = candidate({
+ records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) =>
+ fixture === "reference-250" && stage === "commandQueryMs"
+ ? {
+ fixture,
+ stage,
+ runs: [
+ [...Array(15).fill(10)],
+ [...Array(15).fill(500)],
+ [...Array(15).fill(500)]
+ ]
+ }
+ : { fixture, stage, runs: [samples(10), samples(11), samples(12)] }
+ )
+ });
+ expect(() => assertTimeToAnswerPromotion(artifact(), slowMajority)).toThrow("promotion limit");
+ });
+
+ it("uses phase-normalized p95 as the controlled stage-5 promotion authority", () => {
+ const stageFiveRuns = (repeatedValue: number) =>
+ Array.from({ length: TIME_TO_ANSWER_CONTROLLED_RUNS }, () =>
+ Array.from({ length: TIME_TO_ANSWER_WARMED_SAMPLES }, (_, index) => (index < 10 ? 10 : repeatedValue))
+ );
+ const baseline = artifact({
+ records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) =>
+ fixture === "reference-25" && stage === "publicationToNodeObservationMs"
+ ? { fixture, stage, runs: stageFiveRuns(10) }
+ : { fixture, stage, runs: [samples(10), samples(11), samples(12)] }
+ )
+ });
+ const candidateStageFiveSlow = candidate({
+ records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) =>
+ fixture === "reference-25" && stage === "publicationToNodeObservationMs"
+ ? { fixture, stage, runs: stageFiveRuns(100) }
+ : { fixture, stage, runs: [samples(10), samples(11), samples(12)] }
+ )
+ });
+ // Ordinary per-run p95 is 100 in both artifacts, but all-repeat phase medians
+ // make the normalized figure rise from 10 to 55 and reject promotion.
+ expect(() => assertTimeToAnswerPromotion(baseline, candidateStageFiveSlow)).toThrow("promotion limit");
+ });
+
+ it("cannot let a cold first observation enter a warmed stage summary", () => {
+ // The daemon's first passes are cold, and with 15 recorded samples nearest-rank
+ // p95 IS the maximum — so a surviving cold observation would become the
+ // published number. The protocol discards a FIXED 60 observations (declared
+ // before the capture), keeps them all for audit, and never trims further.
+ const cold = Array.from({ length: TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS }, (_, index) => 99.9 - index);
+ const warm = Array.from({ length: TIME_TO_ANSWER_WARMED_SAMPLES }, (_, index) => 2 + index * 0.1);
+ const { warmUps, recorded } = splitWarmedObservations([...cold, ...warm]);
+ expect(warmUps).toEqual(cold);
+ expect(warmUps).toHaveLength(cold.length);
+ expect(recorded).toEqual(warm);
+ const summary = summarizeTimeToAnswerStage([recorded, recorded, recorded]);
+ expect(summary.runP95Ms.every((value) => value < 10)).toBe(true);
+ expect(summary.medianOfThreeRunP95Ms).toBeLessThan(10);
+ // No arbitrary sampling: the whole window is required and a short window FAILS
+ // rather than being silently trimmed to the declared count.
+ expect(() => splitWarmedObservations([...cold, ...warm].slice(0, cold.length + warm.length - 1))).toThrow();
+ });
+
+ it("never adapts the measured window to diagnostics", () => {
+ const observations = Array.from({ length: 75 }, (_, index) => index < 60 ? 100 - index : 2);
+ const { warmUps, recorded } = splitWarmedObservations(observations);
+ expect(warmUps).toHaveLength(60);
+ expect(recorded).toEqual(Array(15).fill(2));
+ });
+
+ it("binds the executed daemon binary to the recorded revision", () => {
+ const digest = "a".repeat(64);
+ expect(() =>
+ assertDaemonBinaryProvenance({ expectedSha256: digest, observedSha256: digest, phase: "before" })
+ ).not.toThrow();
+ expect(() =>
+ assertDaemonBinaryProvenance({
+ expectedSha256: digest,
+ observedSha256: "b".repeat(64),
+ phase: "before capture"
+ })
+ ).toThrow("daemon binary provenance failed");
+ expect(() =>
+ assertDaemonBinaryProvenance({ expectedSha256: "not-a-digest", observedSha256: digest, phase: "before" })
+ ).toThrow("two lowercase sha256 digests");
+ });
+
+ it("classifies every stage as cold-start, warmed-repeated or scenario-specific", () => {
+ for (const stage of TIME_TO_ANSWER_STAGES) {
+ expect(TIME_TO_ANSWER_STAGE_KIND[stage.id]).toBeDefined();
+ }
+ // Process start is genuinely cold: its first observation IS the measurement.
+ expect(TIME_TO_ANSWER_STAGE_KIND.daemonStartToListenerMs).toBe("cold-start");
+ expect(TIME_TO_ANSWER_STAGE_KIND.listenerToFirstDockerModelMs).toBe("cold-start");
+ // The daemon-side attribution stages are warmed repeated operations.
+ expect(TIME_TO_ANSWER_STAGE_KIND.dockerObservationMs).toBe("warmed-repeated");
+ expect(TIME_TO_ANSWER_STAGE_KIND.composeEnrichmentMs).toBe("warmed-repeated");
+ expect(TIME_TO_ANSWER_STAGE_KIND.findingsDerivationMs).toBe("warmed-repeated");
+ // Scenario cells are declared only for scenario fixtures.
+ expect(isScenarioCell("slow-bounded-compose-projection", "composeEnrichmentMs")).toBe(true);
+ expect(isScenarioCell("reference-25", "composeEnrichmentMs")).toBe(false);
+ });
+
+ it("rejects a candidate with different browser flags", () => {
+ expect(() =>
+ assertTimeToAnswerPromotion(
+ artifact(),
+ candidate({ environment: { browserFlags: ["--disable-background-networking", "--enable-gpu"] } })
+ )
+ ).toThrow("does not match the pinned baseline environment");
+ });
+
+ it("rejects a candidate built in a non-production mode", () => {
+ expect(() => validateTimeToAnswerEvidence(candidate({ environment: { buildMode: "development" } }))).toThrow(
+ "closed safe metadata fields"
+ );
+ });
+
+ it("rejects evidence with a missing stage", () => {
+ const records = TIME_TO_ANSWER_MATRIX.slice(1).map(({ fixture, stage }) => ({
+ fixture,
+ stage,
+ runs: [samples(10), samples(11), samples(12)]
+ }));
+ expect(() => validateTimeToAnswerEvidence(artifact({ records }))).toThrow("exact fixture × stage matrix");
+ });
+
+ it("rejects evidence with an undeclared stage", () => {
+ const records = [
+ ...TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => ({
+ fixture,
+ stage,
+ runs: [samples(10), samples(11), samples(12)]
+ })),
+ { fixture: "reference-25", stage: "inventedStageMs", runs: [samples(1), samples(1), samples(1)] }
+ ];
+ expect(() => validateTimeToAnswerEvidence(artifact({ records }))).toThrow("exact fixture × stage matrix");
+ });
+
+ it("rejects a malformed sample count", () => {
+ const records = TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) =>
+ fixture === "reference-25" && stage === "commandQueryMs"
+ ? { fixture, stage, runs: [samples(10).slice(0, 14), samples(11), samples(12)] }
+ : { fixture, stage, runs: [samples(10), samples(11), samples(12)] }
+ );
+ expect(() => validateTimeToAnswerEvidence(artifact({ records }))).toThrow(
+ `requires exactly ${TIME_TO_ANSWER_WARMED_SAMPLES} finite`
+ );
+ });
+
+ it("rejects a stage with too few controlled runs", () => {
+ const records = TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) =>
+ fixture === "reference-100" && stage === "buildModelMs"
+ ? { fixture, stage, runs: [samples(10), samples(11)] }
+ : { fixture, stage, runs: [samples(10), samples(11), samples(12)] }
+ );
+ expect(() => validateTimeToAnswerEvidence(artifact({ records }))).toThrow(
+ "requires exactly three raw runs per stage"
+ );
+ expect(TIME_TO_ANSWER_CONTROLLED_RUNS).toBe(3);
+ });
+
+ it("rejects a supplied or fabricated summary instead of recomputing it", () => {
+ const records = TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) =>
+ fixture === "reference-250" && stage === "composeEnrichmentMs"
+ ? {
+ fixture,
+ stage,
+ runs: [samples(10), samples(11), samples(12)],
+ summary: { runP95Ms: [1, 1, 1], medianOfThreeRunP95Ms: 1 }
+ }
+ : { fixture, stage, runs: [samples(10), samples(11), samples(12)] }
+ );
+ expect(() => validateTimeToAnswerEvidence(artifact({ records }))).toThrow("unsafe or incomplete shape");
+ });
+
+ it("rejects a negative or non-finite sample", () => {
+ for (const bad of [-1, Number.NaN, Number.POSITIVE_INFINITY, "12" as unknown as number]) {
+ const records = TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) =>
+ fixture === "reference-25" && stage === "findingsDerivationMs"
+ ? { fixture, stage, runs: [[bad, ...samples(10).slice(1)], samples(11), samples(12)] }
+ : { fixture, stage, runs: [samples(10), samples(11), samples(12)] }
+ );
+ expect(() => validateTimeToAnswerEvidence(artifact({ records }))).toThrow();
+ }
+ });
+
+ it("rejects an unknown baseline identifier", () => {
+ expect(() => validateTimeToAnswerEvidence({ ...artifact(), baseline: "dockermap-v1/other" })).toThrow(
+ "closed baseline/environment/records schema"
+ );
+ });
+});
diff --git a/apps/web/vite.config.ts b/apps/web/vite.config.ts
index 0ea2ae44..6ed5fdae 100644
--- a/apps/web/vite.config.ts
+++ b/apps/web/vite.config.ts
@@ -3,6 +3,16 @@ import react from "@vitejs/plugin-react";
export default defineConfig({
plugins: [react()],
+ /**
+ * The benchmark instrumentation seam (#335) is compile-time gated. The
+ * production build defines the flag `false`, so the acceptance hook, its event
+ * identifiers, the probe entry and the artificial render-delay machinery are
+ * all removed by dead-code elimination before minification. Flipping this value
+ * here is exactly what must never happen: tests/perf/productionIsolation.test.mjs
+ * reads this file and requires the production definition to be `false`, and
+ * inspects the built artifact for the seam's identifiers.
+ */
+ define: { __DOCKERMAP_BENCH_ACCEPTANCE__: "false" },
server: {
port: 3233
}
diff --git a/crates/dockermap-daemon/src/bench_timing.rs b/crates/dockermap-daemon/src/bench_timing.rs
new file mode 100644
index 00000000..05152440
--- /dev/null
+++ b/crates/dockermap-daemon/src/bench_timing.rs
@@ -0,0 +1,156 @@
+//! Test-only stage attribution for the time-to-answer benchmark (#335).
+//!
+//! This module measures the *current* implementation without changing it. The
+//! Docker observation and the Compose filesystem projection are timed as two
+//! separately attributable stages even though they still execute inside one
+//! publication budget; #336 owns moving the projection off that critical path.
+//!
+//! It is inert unless `DOCKERMAP_BENCH_STAGE_TIMING_PATH` names an absolute
+//! path. When unset or unusable, nothing is measured, nothing is written, and
+//! no public response changes. There is no route, no response field, and no
+//! production telemetry: the sink is an append-only newline-delimited JSON file
+//! chosen by the benchmark harness.
+
+use std::io::Write;
+use std::path::{Path, PathBuf};
+use std::time::Instant;
+
+const BENCH_STAGE_TIMING_PATH_ENV: &str = "DOCKERMAP_BENCH_STAGE_TIMING_PATH";
+
+/// Docker inventory observation (containers, networks, volumes).
+pub(crate) const STAGE_DOCKER_OBSERVATION: &str = "dockerObservationMs";
+/// Compose filesystem projection for the same publication.
+pub(crate) const STAGE_COMPOSE_ENRICHMENT: &str = "composeEnrichmentMs";
+/// Findings derivation for the published runtime map.
+pub(crate) const STAGE_FINDINGS_DERIVATION: &str = "findingsDerivationMs";
+
+/// Resolve the configured sink. Only an absolute path is accepted, so a
+/// relative value cannot silently land inside a working directory, and an
+/// empty or malformed value disables the hook instead of failing a publication.
+pub(crate) fn sink_from_env_value(value: Option) -> Option {
+ let value = value?;
+ let trimmed = value.trim();
+ if trimmed.is_empty() {
+ return None;
+ }
+ let path = Path::new(trimmed);
+ if !path.is_absolute() {
+ return None;
+ }
+ Some(path.to_path_buf())
+}
+
+pub(crate) fn sink() -> Option {
+ sink_from_env_value(std::env::var(BENCH_STAGE_TIMING_PATH_ENV).ok())
+}
+
+/// One NDJSON record. Kept pure so its shape is testable without touching the
+/// process environment or the filesystem.
+pub(crate) fn stage_timing_line(stage: &str, milliseconds: f64) -> String {
+ format!("{{\"stage\":\"{stage}\",\"ms\":{milliseconds:.3}}}\n")
+}
+
+fn write_line(sink: Option<&Path>, line: &str) {
+ let Some(path) = sink else {
+ return;
+ };
+ // A benchmark sink must never be able to interrupt collection: a write
+ // failure is dropped, not propagated.
+ if let Ok(mut file) = std::fs::OpenOptions::new()
+ .create(true)
+ .append(true)
+ .open(path)
+ {
+ let _ = file.write_all(line.as_bytes());
+ }
+}
+
+/// Record one stage duration. A no-op when the hook is disabled.
+pub(crate) fn record(sink: Option<&Path>, stage: &str, start: Instant) {
+ if sink.is_none() {
+ return;
+ }
+ let elapsed = start.elapsed();
+ write_line(
+ sink,
+ &stage_timing_line(stage, elapsed.as_secs_f64() * 1000.0),
+ );
+}
+
+#[cfg(test)]
+mod tests {
+ use super::*;
+ use std::time::Duration;
+
+ #[test]
+ fn sink_requires_an_absolute_non_empty_path() {
+ assert!(sink_from_env_value(None).is_none());
+ assert!(sink_from_env_value(Some(String::new())).is_none());
+ assert!(sink_from_env_value(Some(" ".into())).is_none());
+ assert!(sink_from_env_value(Some("relative/bench.jsonl".into())).is_none());
+ assert!(sink_from_env_value(Some("./bench.jsonl".into())).is_none());
+ assert_eq!(
+ sink_from_env_value(Some("/tmp/dockermap-bench.jsonl".into())).as_deref(),
+ Some(Path::new("/tmp/dockermap-bench.jsonl"))
+ );
+ assert_eq!(
+ sink_from_env_value(Some(" /tmp/dockermap-bench.jsonl ".into())).as_deref(),
+ Some(Path::new("/tmp/dockermap-bench.jsonl"))
+ );
+ }
+
+ #[test]
+ fn stage_lines_are_closed_newline_delimited_json() {
+ assert_eq!(
+ stage_timing_line(STAGE_DOCKER_OBSERVATION, 12.3456),
+ "{\"stage\":\"dockerObservationMs\",\"ms\":12.346}\n"
+ );
+ assert_eq!(
+ stage_timing_line(STAGE_COMPOSE_ENRICHMENT, 0.0),
+ "{\"stage\":\"composeEnrichmentMs\",\"ms\":0.000}\n"
+ );
+ assert!(stage_timing_line(STAGE_DOCKER_OBSERVATION, 1.0).ends_with('\n'));
+ }
+
+ #[test]
+ fn disabled_hook_writes_nothing() {
+ let directory = tempfile::tempdir().expect("temporary bench directory");
+ let target = directory.path().join("bench.jsonl");
+ record(None, STAGE_DOCKER_OBSERVATION, Instant::now());
+ for _ in 0..3 {
+ record(None, STAGE_COMPOSE_ENRICHMENT, Instant::now());
+ }
+ assert!(
+ !target.exists(),
+ "a disabled bench hook must not create a sink file"
+ );
+ assert!(std::fs::read_dir(directory.path())
+ .expect("bench directory")
+ .next()
+ .is_none());
+ }
+
+ #[test]
+ fn enabled_hook_appends_one_line_per_stage() {
+ let directory = tempfile::tempdir().expect("temporary bench directory");
+ let target = directory.path().join("bench.jsonl");
+ let start = Instant::now();
+ std::thread::sleep(Duration::from_millis(1));
+ record(Some(&target), STAGE_DOCKER_OBSERVATION, start);
+ record(Some(&target), STAGE_COMPOSE_ENRICHMENT, start);
+ let written = std::fs::read_to_string(&target).expect("bench sink is readable");
+ let lines = written.lines().collect::>();
+ assert_eq!(lines.len(), 2);
+ assert!(lines[0].starts_with("{\"stage\":\"dockerObservationMs\",\"ms\":"));
+ assert!(lines[1].starts_with("{\"stage\":\"composeEnrichmentMs\",\"ms\":"));
+ for line in lines {
+ assert!(line.ends_with('}'));
+ let value: serde_json::Value = serde_json::from_str(line).expect("valid JSON line");
+ let ms = value
+ .get("ms")
+ .and_then(|ms| ms.as_f64())
+ .expect("numeric ms");
+ assert!(ms.is_finite() && ms >= 0.0);
+ }
+ }
+}
diff --git a/crates/dockermap-daemon/src/cache_refresh.rs b/crates/dockermap-daemon/src/cache_refresh.rs
index 7970a919..df3422a3 100644
--- a/crates/dockermap-daemon/src/cache_refresh.rs
+++ b/crates/dockermap-daemon/src/cache_refresh.rs
@@ -638,6 +638,8 @@ impl DaemonCache {
.assign(&mut self.snapshot, &mut self.health, &mut self.runtime_map);
// Findings are a pure projection of the sanitized runtime map, so
// calculate and cache them only after the publication revision exists.
+ let bench_sink = crate::bench_timing::sink();
+ let findings_started = std::time::Instant::now();
let mut findings = derive_findings(&self.runtime_map);
if self.health.mode == RuntimeMode::Docker {
if let Some((scan, binding)) = &self.compose_runtime_binding {
@@ -647,6 +649,11 @@ impl DaemonCache {
}
}
findings.sort_by(|left, right| left.id.cmp(&right.id));
+ crate::bench_timing::record(
+ bench_sink.as_deref(),
+ crate::bench_timing::STAGE_FINDINGS_DERIVATION,
+ findings_started,
+ );
self.findings = FindingsResponse {
summary: FindingSummary::from_findings(&findings),
findings,
@@ -863,12 +870,22 @@ where
+ 'static,
{
let started = tokio::time::Instant::now();
+ // Test-only stage attribution (#335). Disabled unless the benchmark harness
+ // sets an absolute DOCKERMAP_BENCH_STAGE_TIMING_PATH; it records durations
+ // only and never changes what is collected or published.
+ let bench_sink = crate::bench_timing::sink();
+ let bench_docker_started = std::time::Instant::now();
let observation =
match tokio::time::timeout(snapshot_timeout, collector.collect_observation()).await {
Ok(Ok(observation)) => observation,
Ok(Err(error)) => return Err(DockerReadFailure::Failed(error)),
Err(_) => return Err(DockerReadFailure::TimedOut),
};
+ crate::bench_timing::record(
+ bench_sink.as_deref(),
+ crate::bench_timing::STAGE_DOCKER_OBSERVATION,
+ bench_docker_started,
+ );
let mut snapshot = observation.snapshot;
snapshot.images = derive_images(&snapshot);
let collected_at = snapshot.last_updated;
@@ -885,10 +902,21 @@ where
let _flight = flight;
projection(observation.compose_containers, collected_at)
});
- match tokio::time::timeout(remaining, projection_task).await {
+ let projection_started = std::time::Instant::now();
+ let binding = match tokio::time::timeout(remaining, projection_task).await {
Ok(Ok(binding)) => binding,
Ok(Err(_)) | Err(_) => None,
- }
+ };
+ // Attributed separately from the Docker observation so the
+ // baseline can show that this projection currently sits inside
+ // the Docker publication budget. #336 owns moving it off that
+ // path; nothing is decoupled here.
+ crate::bench_timing::record(
+ bench_sink.as_deref(),
+ crate::bench_timing::STAGE_COMPOSE_ENRICHMENT,
+ projection_started,
+ );
+ binding
} else {
None
}
diff --git a/crates/dockermap-daemon/src/main.rs b/crates/dockermap-daemon/src/main.rs
index 7d0b5e52..a309054c 100644
--- a/crates/dockermap-daemon/src/main.rs
+++ b/crates/dockermap-daemon/src/main.rs
@@ -1,4 +1,5 @@
mod auth;
+mod bench_timing;
mod cache_refresh;
mod compose_api;
mod config;
diff --git a/docs/testing/TIME_TO_ANSWER_BASELINE.md b/docs/testing/TIME_TO_ANSWER_BASELINE.md
new file mode 100644
index 00000000..cb793652
--- /dev/null
+++ b/docs/testing/TIME_TO_ANSWER_BASELINE.md
@@ -0,0 +1,202 @@
+# Time-to-answer Baseline 4
+
+Status: **accepted candidate under review in PR #346**. Baseline 4 is the intended
+comparison authority for #336 and #337 once that PR is merged. It is not represented
+as merged or as an unconditional authority before review completes.
+
+- baseline id / methodology: `dockermap-v1/time-to-answer-methodology-8`
+- product and capture-harness revision: `bdce6ae354d757d3318c514e10d620edb497918e`
+- durable evidence root: `/srv/jonas/evidence/dockermap/time-to-answer/bdce6ae/`
+ (stored outside this repository under the evidence-artifact policy)
+- general raw artifact: `time-to-answer-baseline-4.json`, sha256
+ `916610bdd4767fb01a4f57e29f318aa504836875cbb4d87ba615c70cb6fb7300`
+- capture duration: 105.1 minutes; 44 records × 3 runs × 15 measured samples =
+ 1,980 raw samples
+
+Baselines 1, 2, and 3 are **REJECTED** historical attempts and are not current
+figures or authority for any comparison. Baseline 1 used an uncommitted harness and
+incorrect stage boundaries; Baseline 2 retained cold data and shared a stage clock;
+Baseline 3 did not sweep poll phase, retained an insufficient warm-up, and had
+incorrect provenance/compatibility handling. Their measurement numbers do not appear
+in this record.
+
+## Conditioning and composite contract
+
+Every ordinary warmed end-to-end cell uses exactly 60 fixed burn-in observations
+followed by exactly 15 measured observations. Observation 61 is always the first
+measured sample. Burn-in is retained for audit but excluded completely from timing
+summaries and promotion comparisons. This is fixed, deterministic conditioning for
+baseline and candidate; it makes **no stationarity claim**.
+
+The composite authority has 44 rows: 38 `end-to-end` rows from
+`time-to-answer-baseline-4.json`, plus 6 `controlled-poll-phase` rows from
+`time-to-answer-stage5.json`. Every row records fixture, stage, measurement protocol,
+source evidence file, and the checkpoint above. All 44 composite runs equal their raw
+source runs; there are no duplicate cells.
+
+Normal Stage-6 and Stage-7 timing rows are ordinary `end-to-end` measurements.
+The separate `controlled-stage6-stage7-seam-isolation` control is supporting evidence
+only and is **FAIL** at this checkpoint:
+
+```
+[independence] FAIL: Error: control delay did not begin after the acceptance timestamp
+```
+
+Its disclosed limitation is `publication-level causal identity unavailable` and
+`validatesDaemonToBrowserAttribution: false`. It is not a timing row, its samples are
+not Baseline-4 timing observations, and it is not daemon→browser attribution evidence.
+
+## Recomputed timing table
+
+This table is copied from the recomputed summary
+`time-to-answer-baseline-4-summary.md`, derived from the composite raw sample arrays.
+
+| fixture | stage | run p95 (ms) | reviewed aggregation | reviewed (ms) | min | max |
+| --- | --- | --- | --- | --- | --- | --- |
+| reference-25 | daemonStartToListenerMs | 38.22 / 37.28 / 41.97 | median-of-three-run-p95 | 38.22 | 33.77 | 41.97 |
+| reference-25 | listenerToFirstDockerModelMs | 11.86 / 11.07 / 11.50 | median-of-three-run-p95 | 11.50 | 9.13 | 11.86 |
+| reference-25 | dockerObservationMs | 1.74 / 1.71 / 1.84 | median-of-three-run-p95 | 1.74 | 1.19 | 1.84 |
+| reference-25 | composeEnrichmentMs | 1.01 / 1.06 / 1.02 | median-of-three-run-p95 | 1.02 | 0.62 | 1.06 |
+| reference-25 | notificationToCoherentModelMs | 21.20 / 22.40 / 18.00 | median-of-three-run-p95 | 21.20 | 11.20 | 22.40 |
+| reference-25 | coherentModelToUsefulRenderMs | 51.00 / 37.60 / 30.20 | median-of-three-run-p95 | 37.60 | 17.60 | 51.00 |
+| reference-25 | buildModelMs | 0.20 / 0.20 / 0.20 | median-of-three-run-p95 | 0.20 | 0.00 | 0.20 |
+| reference-25 | findingsDerivationMs | 0.08 / 0.06 / 0.08 | median-of-three-run-p95 | 0.08 | 0.04 | 0.08 |
+| reference-25 | legacyTopologyLayoutMs | 1.90 / 1.90 / 1.90 | median-of-three-run-p95 | 1.90 | 1.60 | 1.90 |
+| reference-25 | commandQueryMs | 52.60 / 55.50 / 59.90 | median-of-three-run-p95 | 55.50 | 6.00 | 59.90 |
+| reference-25 | productionBundleMs | 50.30 / 54.70 / 58.60 | median-of-three-run-p95 | 54.70 | 37.80 | 58.60 |
+| reference-25 | publicationToNodeObservationMs | 1897.38 / 1898.20 / 1898.67 | phase-normalized-p95 | 1897.79 | 97.18 | 1898.67 |
+| reference-100 | daemonStartToListenerMs | 118.14 / 129.97 / 110.31 | median-of-three-run-p95 | 118.14 | 99.22 | 129.97 |
+| reference-100 | listenerToFirstDockerModelMs | 22.30 / 18.32 / 19.11 | median-of-three-run-p95 | 19.11 | 15.23 | 22.30 |
+| reference-100 | dockerObservationMs | 3.24 / 4.34 / 2.79 | median-of-three-run-p95 | 3.24 | 2.16 | 4.34 |
+| reference-100 | composeEnrichmentMs | 1.07 / 1.01 / 1.05 | median-of-three-run-p95 | 1.05 | 0.63 | 1.07 |
+| reference-100 | notificationToCoherentModelMs | 35.90 / 37.20 / 45.80 | median-of-three-run-p95 | 37.20 | 29.10 | 45.80 |
+| reference-100 | coherentModelToUsefulRenderMs | 52.40 / 53.80 / 48.20 | median-of-three-run-p95 | 52.40 | 35.10 | 53.80 |
+| reference-100 | buildModelMs | 0.50 / 0.50 / 0.40 | median-of-three-run-p95 | 0.50 | 0.20 | 0.50 |
+| reference-100 | findingsDerivationMs | 0.21 / 0.23 / 0.26 | median-of-three-run-p95 | 0.23 | 0.16 | 0.26 |
+| reference-100 | legacyTopologyLayoutMs | 28.30 / 23.10 / 24.40 | median-of-three-run-p95 | 24.40 | 19.40 | 28.30 |
+| reference-100 | commandQueryMs | 18.20 / 40.10 / 22.80 | median-of-three-run-p95 | 22.80 | 8.20 | 40.10 |
+| reference-100 | productionBundleMs | 52.00 / 50.50 / 61.10 | median-of-three-run-p95 | 52.00 | 37.20 | 61.10 |
+| reference-100 | publicationToNodeObservationMs | 1898.07 / 1898.53 / 1897.64 | phase-normalized-p95 | 1897.61 | 98.18 | 1898.53 |
+| reference-250 | daemonStartToListenerMs | 311.26 / 325.69 / 308.36 | median-of-three-run-p95 | 311.26 | 106.85 | 325.69 |
+| reference-250 | listenerToFirstDockerModelMs | 170.76 / 123.58 / 121.55 | median-of-three-run-p95 | 123.58 | 42.40 | 170.76 |
+| reference-250 | dockerObservationMs | 7.26 / 6.21 / 7.46 | median-of-three-run-p95 | 7.26 | 4.42 | 7.46 |
+| reference-250 | composeEnrichmentMs | 0.94 / 0.97 / 1.14 | median-of-three-run-p95 | 0.97 | 0.63 | 1.14 |
+| reference-250 | notificationToCoherentModelMs | 91.00 / 91.30 / 90.90 | median-of-three-run-p95 | 91.00 | 70.30 | 91.30 |
+| reference-250 | coherentModelToUsefulRenderMs | 188.00 / 191.00 / 188.70 | median-of-three-run-p95 | 188.70 | 145.40 | 191.00 |
+| reference-250 | buildModelMs | 1.40 / 1.70 / 1.90 | median-of-three-run-p95 | 1.70 | 0.60 | 1.90 |
+| reference-250 | findingsDerivationMs | 0.81 / 1.03 / 0.80 | median-of-three-run-p95 | 0.81 | 0.58 | 1.03 |
+| reference-250 | legacyTopologyLayoutMs | 150.40 / 166.00 / 149.10 | median-of-three-run-p95 | 150.40 | 123.30 | 166.00 |
+| reference-250 | commandQueryMs | 26.10 / 28.70 / 28.40 | median-of-three-run-p95 | 28.40 | 11.40 | 28.70 |
+| reference-250 | productionBundleMs | 52.40 / 60.40 / 54.70 | median-of-three-run-p95 | 54.70 | 40.90 | 60.40 |
+| reference-250 | publicationToNodeObservationMs | 1898.24 / 1897.54 / 1898.61 | phase-normalized-p95 | 1897.67 | 96.58 | 1898.61 |
+| slow-bounded-compose-projection | composeEnrichmentMs | 6.62 / 9.55 / 11.77 | median-of-three-run-p95 | 9.55 | 5.13 | 11.77 |
+| provider-only-revision-change | notificationToCoherentModelMs | 45.50 / 41.30 / 36.60 | median-of-three-run-p95 | 41.30 | 27.80 | 45.50 |
+| provider-only-revision-change | publicationToNodeObservationMs | 1896.43 / 1898.67 / 1898.03 | phase-normalized-p95 | 1898.00 | 97.02 | 1898.67 |
+| docker-topology-change | notificationToCoherentModelMs | 46.60 / 43.70 / 48.30 | median-of-three-run-p95 | 46.60 | 29.40 | 48.30 |
+| docker-topology-change | coherentModelToUsefulRenderMs | 46.70 / 52.40 / 44.50 | median-of-three-run-p95 | 46.70 | 37.30 | 52.40 |
+| docker-topology-change | publicationToNodeObservationMs | 1898.41 / 1897.48 / 1898.77 | phase-normalized-p95 | 1897.88 | 97.46 | 1898.77 |
+| unavailable-optional-provider | notificationToCoherentModelMs | 44.30 / 50.10 / 48.50 | median-of-three-run-p95 | 48.50 | 26.50 | 50.10 |
+| unavailable-optional-provider | publicationToNodeObservationMs | 1898.45 / 1898.15 / 1897.72 | phase-normalized-p95 | 1897.93 | 97.40 | 1898.45 |
+
+## Stage 5 controlled poll-phase results
+
+The dedicated `controlled-poll-phase` protocol owns arm → mark → trigger →
+identity-acknowledgement. It covers all six declared
+`publicationToNodeObservationMs` fixtures on ten declared phases (100 through 1900
+ms) over the 2000 ms poll interval. There are 45 distinct trigger ids per fixture;
+the maximum absolute observed-vs-declared phase error is at most 1.0 ms. The published
+figure is phase-normalized, not user-traffic or network latency.
+
+The recomputed phase table prints these per-phase medians for the reference and
+topology fixtures:
+
+| fixture | declared phases | observed latency median per declared phase (ms, earliest→latest) | phase-normalized p95 (ms) | span (ms) |
+| --- | --- | --- | --- | --- |
+| reference-25 | 10 | 1898, 1699, 1498, 1299, 1098, 899, 698, 498, 298, 97 | 1897.79 | 1801.49 |
+| reference-100 | 10 | 1898, 1698, 1498, 1298, 1098, 898, 698, 498, 298, 99 | 1897.61 | 1800.34 |
+| reference-250 | 10 | 1898, 1698, 1498, 1298, 1098, 898, 698, 499, 299, 97 | 1897.67 | 1802.03 |
+| docker-topology-change | 10 | 1898, 1698, 1498, 1298, 1098, 899, 698, 498, 298, 99 | 1897.88 | 1801.32 |
+
+The phase-normalized figure weights declared phases uniformly. The complete six-row
+reviewed values are in the timing table above; the summary's dedicated phase table is
+the source for the printed per-phase curves.
+
+## Bucket shares
+
+Source: `time-to-answer-baseline-4-summary.md`.
+
+### reference-25 buckets (sum of reviewed stage figures: 2121.45 ms)
+- transport-notification: 1897.79 ms (89.5%)
+- rendering: 94.20 ms (4.4%)
+- search: 55.50 ms (2.6%)
+- backend-collection: 52.56 ms (2.5%)
+- browser-model: 21.40 ms (1.0%)
+
+### reference-100 buckets (sum of reviewed stage figures: 2228.69 ms)
+- transport-notification: 1897.61 ms (85.1%)
+- backend-collection: 141.78 ms (6.4%)
+- rendering: 128.80 ms (5.8%)
+- browser-model: 37.70 ms (1.7%)
+- search: 22.80 ms (1.0%)
+
+### reference-250 buckets (sum of reviewed stage figures: 2856.44 ms)
+- transport-notification: 1897.67 ms (66.4%)
+- backend-collection: 443.88 ms (15.5%)
+- rendering: 393.80 ms (13.8%)
+- browser-model: 92.70 ms (3.2%)
+- search: 28.40 ms (1.0%)
+
+## Burn-in audit
+
+`time-to-answer-baseline-4.json.harness-evidence.json` proves 96 warmed cells with
+complete 75-observation windows: burn-in is observations 1–60, measurement is
+observations 61–75, and there are zero mismatches. Eighteen cold-start cells have no
+burn-in. The retained burn-in data is audit material only and never contributes to a
+timing summary.
+
+## Pinned environment and provenance
+
+| field | value |
+| --- | --- |
+| runnerClass | linux-x86_64-dedicated |
+| cpuClass | cpus-16vcpu |
+| osImage / kernel | ubuntu-26.04 / 7.0.0-31-generic |
+| Node / Rust / Docker | 22.23.2 / 1.88.0 / 29.8.1 |
+| SSE poll interval | 2000 ms |
+| sourceRevision / harnessRevision | `bdce6ae354d757d3318c514e10d620edb497918e` / `bdce6ae354d757d3318c514e10d620edb497918e` |
+| daemon binary SHA-256 | `862a70cac056dcdbfc0593a03050adabc14a5f7a780c3e64873bec905f772807` |
+| daemon build command | `cargo build --release --locked -p dockermap-daemon --manifest-path crates/Cargo.toml` |
+| cargo revision | cargo-1.88.0-873a06493-2025-05-10 |
+| browser engine / revision | chromium / 1.61.0 |
+| browser flags | `--disable-background-networking --disable-sync --no-first-run --no-default-browser-check` |
+| font environment / build mode | system-default / production |
+| fixture revision | dockermap-v1/time-to-answer-fixtures-1 |
+| methodology version | dockermap-v1/time-to-answer-methodology-8 |
+
+The daemon digest was identical before and after capture. The raw sections are
+assembled with:
+
+```
+npx tsx tests/perf/assembleCompositeEvidence.ts --general --stageFive --output
+```
+
+Recompute the published summary with:
+
+```
+npm run perf:summarize -- --artifact
+```
+
+The artifact paths are external evidence-artifact-policy storage, not repository
+deliverables.
+
+## What this baseline does NOT claim
+
+- It is not real-Docker latency: it uses a deterministic fixture daemon.
+- It is not a real network or user-traffic distribution.
+- Stage-5 phase-normalized timing is not network latency and does not claim
+ publications occur uniformly across poll phase.
+- It does not prove the model is complete: Stages 1/2 and Stage 6 end at coherence.
+- It makes no daemon-publication attribution claim from the failed seam-isolation
+ control.
+- It grants no permission to optimise. #336, #337, and #338 must compare in a
+ compatible pinned environment under `max(baseline × 1.25, baseline + 2 ms)`.
diff --git a/docs/testing/TIME_TO_ANSWER_EVIDENCE.md b/docs/testing/TIME_TO_ANSWER_EVIDENCE.md
new file mode 100644
index 00000000..69fcb305
--- /dev/null
+++ b/docs/testing/TIME_TO_ANSWER_EVIDENCE.md
@@ -0,0 +1,699 @@
+# Time-to-answer evidence
+
+## Methodology-8 Baseline-4 conditioning (authoritative)
+
+For every ordinary warmed end-to-end fixture/run, Baseline-4 performs **exactly
+60 fixed burn-in observations**, retains all 60 in harness evidence, then records
+**exactly 15 measured observations**. Observation **61** is always the first
+measured sample. Burn-in is excluded completely from timing summaries and
+promotion comparisons. It is a deterministic, equal conditioning workload for a
+baseline and candidate; it makes no claim that 60 guarantees steady state.
+
+No observed value may infer stationarity, adaptively trim samples, select a
+per-metric warm-up, or extend the burn-in. Stationarity/drift calculations remain
+historical informational diagnostics only and never alter or invalidate an
+otherwise structurally valid ordinary run.
+
+The final 60-observation calibration is retained as **REJECTED,
+NON-AUTHORITATIVE** audit evidence for Baseline-4. It disproved a common sustained
+stationarity validity rule: several metrics stabilized in 2–3 observations,
+`findingsDerivationMs` near 13, `legacyTopologyLayoutMs` near 26,
+`commandQueryMs` near 43, and `buildModelMs` never met the criterion within 60.
+Consequently Baseline-4 does not require calibration PASS, frozen per-metric
+counts, the former +2 margin, or the 0.5–1.5x band.
+
+Controlled evidence is distinct: Stage 5 is `controlled-poll-phase`; Stage-6/7
+seam isolation is `controlled-stage6-stage7-seam-isolation`. Neither contributes
+artificial samples to the normal end-to-end dataset. The normal Stage-6 and
+Stage-7 timings remain ordinary end-to-end 60+15 measurements. A composite is
+incomplete when required Stage-5 controlled evidence is missing; seam-isolation
+evidence is supporting validation rather than a prerequisite for normal timing.
+
+Status: measurement authority for issue #335 and its parent epic #333. This is
+**not** an optimization, a product claim, or permission to cut features for a
+number. Nothing here changes what DockerMap collects or publishes.
+
+DockerMap had controlled performance evidence for the Atlas route only. That is
+not evidence about the operator path, which starts when the daemon process
+starts and ends when a human can act on an answer. This document defines how
+that path is measured, what each number proves, and what it deliberately does
+not.
+
+## The contract lives in code, not in this document
+
+`apps/web/src/lib/performance/timeToAnswerEvidence.ts` is the closed schema and
+the math. It contains no timings. It defines:
+
+- the **12 measured stages**, each with its bucket, a `measures` sentence and a
+ `doesNotProve` sentence;
+- the **fixtures**: `reference-25`, `reference-100`, `reference-250`, plus the
+ scenario fixtures `provider-only-revision-change`, `docker-topology-change`,
+ `slow-bounded-compose-projection` and `unavailable-optional-provider`;
+- the **exact fixture × stage matrix** — **44 cells** — derived from each stage's
+ fixture list, never hand-listed;
+- the **pinned environment allowlist** (runner class, CPU class, OS image and
+ kernel, Node/Rust/Docker revisions, Chromium revision and flags, font
+ environment, production build mode, fixture revision, source revision);
+- **raw-sample validation**: 15 warmed samples in each of 3 complete controlled
+ runs, nearest-rank p95 per run, median of the three run p95 values except that
+ controlled stage 5 is reviewed and promoted by its phase-normalized p95;
+- the **stage-6/7 seam-isolation control** (`assertStageSixSevenIndependence`): a
+ positive artificial presentation delay injected *after* coherent-model
+ acceptance must move stage 7 by at least 70% of that delay and must not move
+ stage 6 beyond `max(30 ms, 25%)`;
+- the **promotion gate** `max(baseline × 1.25, baseline + 2 ms)`, compared only
+ between environments that match on every pinned field except `sourceRevision`
+ (which differs by design) and `dockerRevision` (recorded but informational: no
+ measured stage exercises the host Docker daemon). The other 15 fields must
+ match.
+
+Because summaries are recomputed from the raw samples at review time, a supplied
+summary cannot influence a result. An artifact with a fabricated summary field,
+a truncated matrix, a wrong sample count, a negative or non-numeric sample, an
+unknown stage, a duplicate record, an undeclared fixture, or an extra/unsafe
+metadata field is rejected — see `timeToAnswerEvidence.test.ts`.
+
+## Stage buckets
+
+| bucket | stages | what it answers |
+| --- | --- | --- |
+| backend-collection | daemonStartToListenerMs, listenerToFirstDockerModelMs, dockerObservationMs, composeEnrichmentMs | how long DockerMap takes to have an authoritative answer |
+| transport-notification | publicationToNodeObservationMs | how long a published revision takes to become visible |
+| browser-model | notificationToCoherentModelMs, buildModelMs | how long the browser needs to turn it into a model |
+| rendering | coherentModelToUsefulRenderMs, legacyTopologyLayoutMs, productionBundleMs | how long the operator waits for something useful on screen |
+| search | commandQueryMs | how long a direct question takes to answer |
+
+The buckets exist so the baseline can say **where** the time went — Compose,
+notification, model rebuilds, legacy layout or search — instead of only how much
+there was.
+
+## What each number does and does not prove
+
+The contract carries a `measures`/`doesNotProve` pair for every stage; two
+examples of the distinction that matter most:
+
+- `listenerToFirstDockerModelMs` measures until the first authoritative Docker
+ model is observable. It does **not** prove the model is complete — optional
+ provider evidence may still be missing, and a fast number here must never be
+ read as "the host is fully described".
+- `composeEnrichmentMs` measures Compose filesystem correlation separately from
+ the Docker observation. While the two remain coupled inside one publication
+ budget, this stage is **measured, not removed**; #336 owns moving it off the
+ critical path.
+- `dockerObservationMs` is measured against a deterministic local fixture
+ daemon. It is **not** a claim about a real Docker daemon's latency, host load,
+ or image size.
+- `productionBundleMs` and `commandQueryMs` are measured against the pinned
+ local production build and a fixed representative query set. They are **not**
+ claims about a real network or about operator behaviour.
+
+## Controlled run record
+
+Run exactly three times on the same dedicated pinned runner after a clean
+production build. Record every field of the pinned environment. A missing field,
+a changed fixture/browser/policy/font/runner class, or a runner health failure
+**invalidates the record**; it does not justify retrying until a preferred
+duration appears.
+
+- Keep **raw timings** in the artifact; derive summaries during review.
+- Ordinary unit tests may validate evidence shape and math, and must **not**
+ pretend to be the controlled benchmark. `npm run test:perf` is shape/math and
+ CPU-only; it never compares elapsed production time.
+- Store only sanitized JSON evidence. Never record live host data, raw model or
+ evidence values, credentials, container identities or screenshots.
+- The baseline artifact is an external reviewed record (the same policy the
+ Atlas evidence uses): it is passed to the benchmark job, not committed. The
+ baseline identity (`dockermap-v1/time-to-answer-baseline-1`) and the pinned
+ environment are what are checked in.
+
+## Fixture source
+
+`tests/perf/dockerFixtureTopology.mjs` generates deterministic, secret-free
+Docker inventory for a given container count and scenario, and
+`tests/perf/fake-docker-api.mjs` serves it over a unix socket through the same
+three read-only endpoints the daemon's collector uses
+(`/containers/json`, `/networks`, `/volumes`, plus `/_ping`, `/version`,
+`/info`).
+
+The daemon is pointed at that socket with
+`DOCKERMAP_DOCKER_GATEWAY_SOCKET=` — the same env var it already
+uses for the read-only gateway — so the benchmark exercises the **real**
+collector, projection and publication path. It never contacts a real Docker
+daemon, and the fixture daemon never reads the host filesystem or the network.
+25/100/250 containers is why a deterministic fixture daemon is required at all:
+inventing 250 real containers on a shared host would not be reproducible.
+
+Verified working end to end: the release-built daemon, pointed at the fixture
+socket with 25 containers, reports `mode: docker`, `dockerReachable: true`, and
+publishes 25 containers / 1 network / 5 volumes with a model revision.
+
+**Generation delta.** The harness can advance a fixture's topology generation
+(`POST /__fixture/topology-generation/`), which changes the container
+identities/labels and stops the first `n` containers. Generation 0 is the pristine
+all-running inventory for every fixture; generation `n` therefore makes the
+product render exactly `n` offline/attention services, which is what the stage-7
+expected-content check asserts (`expectedExitedCount` derives the expectation from
+the same generator the fixture daemon serves). Earlier revisions of this harness
+gave `docker-topology-change` a fixed one-in-three exited mix; that mix is gone,
+because a constant mix cannot discriminate a stale render from a fresh one.
+
+**Scenario premises are asserted, not named.** The capture fails when
+`provider-only-revision-change`'s Docker inventory changes, when the
+`unavailable-optional-provider` fixture's optional provider is fresh, and when the
+`slow-bounded-compose-projection` project does not actually declare its
+`SLOW_COMPOSE_SERVICES = 400` services — an empty or truncated project would
+otherwise record a cell and look like a fast projection. The `docker-topology-change`
+and reference fixtures need no extra premise: the stage-7 expected-content check
+binds them to the generation the harness triggered.
+
+## Promotion rules
+
+A candidate passes only when, in an equivalent controlled environment, **every**
+fixture × stage value is at most `max(baseline × 1.25, baseline + 2 ms)`. Limits
+are derived from the measured baseline, never invented as aspirational absolute
+milliseconds. A candidate that fails the environment check fails closed; it is
+not "close enough".
+
+**Provenance is not compatibility.** The environment records 20 pinned fields, and
+they are used in three different ways:
+
+| use | fields | how it is treated |
+| --- | --- | --- |
+| comparison requirements (16) | `runnerClass`, `cpuClass`, `osImage`, `osKernel`, `nodeRevision`, `rustRevision`, `ssePollIntervalMs`, `daemonBinaryBuild`, `cargoRevision`, `browserEngine`, `browserRevision`, `browserFlags`, `fontEnvironment`, `buildMode`, `fixtureRevision`, `methodologyVersion` | must be identical, or the comparison fails closed |
+| provenance/identity (3) | `sourceRevision`, `harnessRevision`, `daemonBinarySha256` | recorded so the artifact identifies exactly what was measured; NEVER required to match |
+| informational (1) | `dockerRevision` | recorded because it is part of the runner's identity; no measured stage exercises the host Docker daemon |
+
+`daemonBinarySha256` in particular must not gate a comparison: the candidate's
+daemon is **rebuilt from the candidate checkout**, so any legitimate change under
+`crates/` — exactly what #336 does — produces a different digest, and a
+byte-identical digest is not reproducible across a changed `CARGO_HOME`.
+
+No optimization claim in #336/#337/#338 (or later) may be accepted without
+comparing against this baseline under this rule.
+
+## Stage attribution inside the daemon
+
+`dockerObservationMs`, `composeEnrichmentMs` and `findingsDerivationMs` are
+measured by a **test-only** hook in the daemon
+(`crates/dockermap-daemon/src/bench_timing.rs`). It is inert unless
+`DOCKERMAP_BENCH_STAGE_TIMING_PATH` names an absolute path; it then appends
+newline-delimited JSON records to that file. There is no route, no response
+field, no runtime telemetry, and no behaviour change.
+
+The hook times the Docker inventory read and the Compose filesystem projection
+**separately while both still execute inside the same Docker publication
+budget**. This baseline is therefore expected to show that Compose projection
+currently sits inside the Docker critical path. That is the measurement, not a
+fix: **nothing is decoupled here, and #336 owns moving the projection off that
+path** — these are the numbers it must improve against.
+
+## Stage 6 and stage 7: two clocks, and why they cannot be one
+
+The two browser stages answer different questions and must not share a clock.
+
+- **Stage 6 — `notificationToCoherentModelMs`.** Starts when the real stream
+ notifies the browser of a new model revision. Ends when the **application
+ accepts one coherent model** — the instant a fetched snapshot/runtime pair with
+ matching generation, provenance and non-empty model revision becomes the model
+ the UI renders.
+- **Stage 7 — `coherentModelToUsefulRenderMs`.** Starts at the *stage-6
+ timestamp*. Ends when the accepted model's **expected Home content is present**,
+ in a commit the application stamped with that accepted revision, followed by a
+ **bounded render/presentation confirmation**: the probe discovers the commit from
+ an animation-frame loop and then awaits a bounded frame after it, so the
+ end-of-stage segment is one or two frames rather than a fixed number. No sleeps
+ are involved, and the raw audit records the commit→end duration of every sample so
+ the mechanism is checkable rather than asserted.
+
+Stage 6 is observed at the **real application seam**: the acceptance point is
+inside `useSystemModel`, at the moment the composed model is published. It is
+never inferred from a DOM mutation — baseline 2 was rejected precisely because
+its stage 7 was element-for-element identical to stage 6 in all 180 samples.
+
+Stage 6 has two measurement modes, and which one applies is decided by the closed
+matrix, never by the sample:
+
+- **content mode** — for every fixture that also declares stage 7: the sample ends
+ only when the (accepted model, rendered content) pair that carries the expected
+ Home content for the triggered change is observed, so an intermediate
+ publication that moves no Home metric cannot be mis-attributed to the sample.
+- **acceptance-only mode** — for the two provider-state fixtures, whose published
+ revision deliberately carries no inventory change: stage 6 ends at the
+ acceptance instant, and stage 7 is not declared for them (requiring a Home
+ repaint there would be an empty number).
+
+**Expected content, not just any repaint.** The fixture's generation delta stops
+the first `g` containers, so generation `g` renders exactly `g` offline/attention
+services. Stage 7 requires the Home metric region to repaint with that exact value
+for the accepted revision, so a stale render, an unrelated repaint (such as the
+topbar clock) or a render belonging to a different revision cannot end it. The
+probe records the pre-change metric value as it arms, so "the DOM changed" is
+measured rather than assumed.
+
+**The chain of custody is checked.** For every browser sample the harness records
+which paired API fetch delivered the accepted revision and which notification
+preceded that fetch cycle, so stage 6 starts at a notification that provably caused
+the fetch the model came from. It does **not** require the accepted revision to
+equal a revision this harness's own stream announced: the daemon is read per
+request, so `/daemon/health` (what the stream carries) and `/daemon/snapshot` (what
+the accepted pair carries) can hold different revisions while a host is churning,
+and every SSE connection polls on its own phase. The overlap with the harness's own
+stream is recorded as evidence (`acceptedRevisionInApiStream`), and for every cell
+that declares stage 7 the accepted revision is additionally bound to the fixture's
+triggered generation by the expected-content check.
+
+## Benchmark-mode application build and production isolation
+
+Stage 6 needs a signal that only exists in application code, so the seam is
+**real product source** (`apps/web/src/lib/performance/modelAcceptance.tsx`) and
+the build decides whether it exists:
+
+| build | flag | what it contains |
+| --- | --- | --- |
+| production (`apps/web/vite.config.ts`) | `__DOCKERMAP_BENCH_ACCEPTANCE__ = "false"` | no seam, no event identifier, no probe entry, no delay machinery |
+| benchmark mode (`tests/perf/benchAppVite.config.mjs`) | `__DOCKERMAP_BENCH_ACCEPTANCE__ = "true"` | the same real app **with** the acceptance seam |
+| benchmark probe (`tests/perf/benchVite.config.mjs`) | — | stages 8 and 10 (real `buildModel`/`layout` modules, in Chromium) |
+
+The application source is never copied or forked: the same files are built twice.
+The product build eliminates every benchmark branch by dead-code elimination
+before minification, and that is asserted against the built artifact — the
+production bundle must not contain `__dockermapBenchAcceptanceSink`,
+`__dockermapBenchRenderDelayMs`, `dockermapAcceptedRevision` or any harness
+identifier (`tests/perf/productionIsolation.test.mjs`), and the capture refuses to
+run at all unless the benchmark-mode build carries the seam and the production
+build does not (`assertBuildIsolation`). The production Vite config must define
+the flag as the literal `"false"`, which the same suite checks by reading it.
+
+Which build serves which stage: stages **6 and 7** are measured on the
+benchmark-mode application build, **stage 11 (Cmd-K)** and **stage 12 (production
+bundle/startup)** on the ordinary production build, and **stages 8 and 10** on the
+benchmark-only module probe. The seam emits only an opaque timestamp plus the
+model revision token, into an in-memory page sink, and stamps the same opaque
+token on the document root so a DOM repaint can be attributed to a revision.
+There is no product payload, no network call, no telemetry and no analytics, and
+the seam adds no route, no API field and no public schema.
+
+### Diagnostics-only layer capture
+
+The benchmark-mode application also keeps a bounded, in-page diagnostic record
+for every coherent model it accepts. The capture harness drains those records to
+`layers.jsonl`, alongside `lifecycle.jsonl` for browser, context, page,
+navigation, teardown, exception, and closure events. Set
+`DOCKERMAP_BENCH_DIAG_DIR` to choose the directory; otherwise they are written
+to the capture raw directory (or beside the raw capture output). These JSONL
+files are observational only: they are not artifact fields, inputs to timing,
+warm-up, stationarity, independence, or promotion rules, and an append failure
+cannot change a measurement result. Both the diagnostic identifiers and the
+in-page sink are compiled out of the ordinary production web bundle.
+
+## Stage 6/7 controlled seam isolation
+
+Seam isolation is its own dedicated supporting protocol (`npm run perf:independence`,
+`tests/perf/captureIndependence.ts`) and is not part of the general end-to-end capture.
+For each fixture that declares both stages it runs the normal three controlled runs of
+15 measured samples, then `TIME_TO_ANSWER_INDEPENDENCE_SAMPLES` (3) control samples in
+which `__dockermapBenchRenderDelayMs = TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS` (250 ms)
+withholds a *newly accepted* publication from the render tree — an artificial
+presentation delay injected **after** acceptance. It creates and tears down its own
+private fixture, daemon, API, benchmark build and browser contexts and refuses partial
+output. The rule enforced before validation:
+
+- **stage 6 must not move** by more than `max(30 ms, 25% of its median)`;
+- **stage 7 must absorb** at least 70% of the injected delay;
+- no control stage-7 sample may be shorter than the injected delay (which would
+ mean the delay never reached the page).
+
+The verdict, the per-run sample sets and the per-sample audit trail are written to the
+protocol's own `--output` artifact and its `--raw-dir` raw record (the raw record is
+retained even when the control fails), and the assembled composite records the verdict in
+`.supporting-evidence.json`. The closed Baseline-4 evidence schema is
+unchanged: the control never becomes artifact timing content and its samples are never
+Baseline-4 timing rows. At this checkpoint the control's status is **FAIL**
+(`[independence] FAIL: Error: control delay did not begin after the acceptance
+timestamp`). The same rule is unit-tested (`timeToAnswerIndependence.test.ts`), including
+the RED cases "the delayed render does not move stage 7" and "stage 6 moves with the
+delayed presentation".
+
+Each control sample arms the browser probe and its one-shot delay before advancing
+the fixture generation. The delay is consumed only after the real
+`useSystemModel` acceptance timestamp, and the accepted snapshot/runtime-map pair
+must have one non-empty matching revision. **Limitation:** the current architecture
+cannot reliably observe publication-level causal identity from the benchmark
+trigger to that accepted pair. This protocol therefore makes no daemon-publication
+attribution claim and does not validate daemon-to-browser attribution; it does not
+replace that limitation with timing proximity, sequence proximity, or matching
+visible content.
+
+## Running the benchmark
+
+```
+# 1. pin the environment from the runner itself
+npm run perf:metadata -- --output /tmp/time-to-answer-metadata.json
+# 2. end-to-end capture of stages 1-4 and 6-12 (3 controlled runs x 15 measured
+# samples per declared cell, after the fixed 60-observation burn-in)
+npm run perf:time-to-answer -- \
+ --metadata /tmp/time-to-answer-metadata.json \
+ --output /tmp/time-to-answer-general.json \
+ --raw-dir /tmp/time-to-answer-raw \
+ --checkpoint
+# 3. dedicated controlled Stage-5 capture; the only owner of the poll-phase protocol
+npm run perf:stage-five -- \
+ --metadata /tmp/time-to-answer-metadata.json \
+ --output /tmp/time-to-answer-stage5.json \
+ --raw-dir /tmp/time-to-answer-stage5-raw
+# 4. dedicated Stage-6/7 seam-isolation control: supporting evidence only, never a
+# Baseline-4 timing row (the companion record states its verdict)
+npm run perf:independence -- \
+ --metadata /tmp/time-to-answer-metadata.json \
+ --output /tmp/stage6-7-seam-isolation.json \
+ --raw-dir /tmp/stage6-7-seam-isolation-raw \
+ --checkpoint
+# 5. assemble the composite authority from the two raw sections
+npx tsx tests/perf/assembleCompositeEvidence.ts \
+ --general /tmp/time-to-answer-general.json \
+ --stageFive /tmp/time-to-answer-stage5.json \
+ --output /tmp/time-to-answer-baseline.json
+# 6. recompute summaries from the raw samples (never trust supplied aggregates)
+npm run perf:summarize -- --artifact /tmp/time-to-answer-baseline.json
+# 7. compare a candidate against a reviewed baseline (fails closed)
+npm run perf:time-to-answer -- \
+ --metadata /tmp/time-to-answer-metadata.json \
+ --output /tmp/time-to-answer-candidate.json \
+ --baseline /tmp/time-to-answer-baseline.json
+```
+
+Prerequisites: a release daemon (`cargo build --release -p dockermap-daemon`),
+Chromium for Playwright, and a built web app — the capture performs the contract,
+production web, benchmark-mode application and module-probe builds itself, and
+pins the artifacts it serves before measuring anything. Each benchmark entrypoint
+owns every process it starts: the general capture owns the fixture Docker daemon,
+the real daemon and API, the production and benchmark builds and real Chromium,
+and the dedicated protocols own their own private fixture, daemon, API, server
+and browser contexts.
+
+**Capture discipline.** The capture refuses to start from a dirty worktree, and
+refuses to run if the metadata's `sourceRevision`, `harnessRevision` or
+`methodologyVersion` does not match the checked-out commit and the contract. A
+baseline is therefore always reproducible from a committed revision: the artifact
+names both the product revision and the harness that measured it, and the design it
+was measured under. Commit the harness **before** capturing — baseline 1 was
+invalidated precisely because its harness existed only as uncommitted changes — and
+run the focused smoke (`DOCKERMAP_BENCH_DEBUG=1` with `--fixtures`, which relaxes
+only the run/sample counts for probing and can never emit an artifact) before
+spending a full capture.
+
+### Warm-up calibration protocol (REJECTED — retained as audit evidence only)
+
+This collector is retained as audit evidence. Per-metric stationarity calibration is
+**retired**: it does not gate Baseline-4, it supplies no warm-up count, and it is not a
+step in the sequence above. The procedure below is recorded so that the retirement is
+auditable, not because it is current.
+
+Calibration is an independent, bounded conditioning collector. It never calls the
+frozen-count lookup, never enters baseline assembly or normal capture's frozen-count
+preflight, and never emits or merges baseline raw evidence. For every
+`warmed-repeated` end-to-end metric and each declared reference fixture
+(`reference-25`, `reference-100`, `reference-250`), it retains exactly **60 ordered
+finite observations** beginning at call zero. This includes daemon-attribution
+metrics, browser/API-path metrics, and the module-probe metrics; the probe's ordinary
+hidden two-call warm-up is disabled for calibration.
+
+For each fixture trace, candidates `w=2..45` compare the median of observations
+`[w-2,w)` with the median of the following 15 observations `[w,w+15)`. A candidate is
+stable only if the ratio is within the frozen **0.5–1.5x** band at that candidate and
+every later eligible candidate. The fixture value is the earliest sustained `w`; the
+metric value is the maximum fixture value plus the frozen safety margin **2**. The
+result must have a complete following 15-observation window within the retained 60.
+A failure is a calibration conflict:
+the collector does not extrapolate, expand the window, retry toward a preferred point,
+or select a fixture-specific baseline count.
+
+### Calibration capacity and superseded evidence
+
+Methodology-8 adopts fixed 60+15 conditioning after the final 60-observation
+calibration disproved the premise that every metric supports one common sustained
+stationarity gate. The calibration capacity, band, and margin are historical
+diagnostic details; they are not a Baseline-4 prerequisite.
+
+The rejected 40-observation calibration remains retained evidence, not Baseline-4
+authority: `buildModelMs: stable w=25 -> frozen warm-up=27 -> requires 42
+observations`. That failure demonstrated insufficient protocol capacity; it does not
+itself define the new window.
+
+The external calibration artifact stores its raw ordered cells, constants, pinned
+environment, daemon-binary provenance, and complete per-metric derivation trace.
+It is persisted with SHA-256 even on conflict, but is **REJECTED,
+NON-AUTHORITATIVE** for Baseline-4: no result table may block or alter capture.
+
+No per-metric warm-up count is published. The calibration results were rejected and
+never supplied a Baseline-4 parameter, so there is no result table to carry forward.
+
+Procedure notes: stages 8 and 10 run against the benchmark-only module probe
+(`tests/perf/benchVite.config.mjs`, real production modules, real Chromium);
+stages 6 and 7 run against the benchmark-mode application build
+(`tests/perf/benchAppVite.config.mjs`); stages 11 and 12 run against the ordinary
+production build. `tests/perf/browserProbe.js` is test-only instrumentation loaded
+before product code. `.bench-dist` and `.bench-app-dist` are generated and
+gitignored. Every capture also writes `