diff --git a/.gitignore b/.gitignore index 41cccf31..43faf39d 100644 --- a/.gitignore +++ b/.gitignore @@ -13,3 +13,5 @@ target .codex/* !.codex/agents/ !.codex/agents/*.toml +tests/perf/.bench-dist +tests/perf/.bench-app-dist diff --git a/apps/web/src/bench-flag.d.ts b/apps/web/src/bench-flag.d.ts new file mode 100644 index 00000000..7679ce20 --- /dev/null +++ b/apps/web/src/bench-flag.d.ts @@ -0,0 +1,12 @@ +/** + * Compile-time flag for the benchmark-only instrumentation seam (#335). + * + * The ordinary production build defines it `false` (apps/web/vite.config.ts), so + * every benchmark branch is removed by dead-code elimination before + * minification. The benchmark-mode application build in `tests/perf` defines it + * `true`. + * + * It is declared here so BOTH builds typecheck against the same product source: + * the seam lives in real application code, never in a copied implementation. + */ +declare const __DOCKERMAP_BENCH_ACCEPTANCE__: boolean; diff --git a/apps/web/src/components/AppShell.tsx b/apps/web/src/components/AppShell.tsx index 75ff01c3..f5ef3f33 100644 --- a/apps/web/src/components/AppShell.tsx +++ b/apps/web/src/components/AppShell.tsx @@ -13,6 +13,7 @@ import { AppContext } from "../context"; import Icon, { type IconName } from "./Icon"; import CommandPalette from "./CommandPalette"; import RouteFocusManager from "./RouteFocusManager"; +import { ModelAcceptanceStamp } from "../lib/performance/modelAcceptance"; import { StateDot, Tag } from "./primitives"; import { UNAVAILABLE_USER } from "../lib/identity"; @@ -275,6 +276,14 @@ export default function AppShell({ onBearerSignOut }: { onBearerSignOut: () => v + {/* + Benchmark-only acceptance stamp (#335). It renders nothing and has no + effect in the product build; the benchmark application build stamps the + accepted revision token in the same commit that renders the accepted + model, so the capture can attribute a DOM repaint to a revision. + */} + + setCommandOpen(false)} model={model} /> ); diff --git a/apps/web/src/hooks/useSystemModel.ts b/apps/web/src/hooks/useSystemModel.ts index 8fb9f24f..2e18f838 100644 --- a/apps/web/src/hooks/useSystemModel.ts +++ b/apps/web/src/hooks/useSystemModel.ts @@ -5,6 +5,7 @@ import { projectRuntimeMap } from "../lib/atlas/project"; import type { AtlasEnvelope } from "../lib/atlas/types"; import type { EvidenceMode, ModelProvenance } from "../lib/evidence"; import { modelProvenanceForMode } from "../lib/evidence"; +import { recordModelAcceptance, recordModelLayers, useDeliveredModel } from "../lib/performance/modelAcceptance"; import { useApiResource } from "./useApiResource"; export interface SystemModelState { @@ -62,6 +63,15 @@ export function useSystemModel(refreshTick: number, evidenceMode: EvidenceMode | const built = buildModel(snapshot.data, runtimeMap.data); lastModel.current = built; lastProvenance.current = snapshot.provenance; + // The acceptance seam (#335). This is the exact point at which a fetched + // resource/revision pair BECOMES the coherent model the UI renders, so it is + // where the benchmark's stage-6 clock starts. It is compile-time gated and + // carries only an opaque timestamp + revision token; see + // lib/performance/modelAcceptance.tsx. +if (__DOCKERMAP_BENCH_ACCEPTANCE__) { + recordModelAcceptance(snapshot.data.modelRevision, runtimeMap.data.modelRevision); +recordModelLayers(snapshot.data, runtimeMap.data, built); +} return built; }, [snapshot.data, snapshot.generation, snapshot.provenance, runtimeMap.data, runtimeMap.generation, runtimeMap.provenance]); @@ -103,10 +113,16 @@ export function useSystemModel(refreshTick: number, evidenceMode: EvidenceMode | }, [snapshot.data, snapshot.generation, snapshot.provenance, runtimeMap.data, runtimeMap.generation, runtimeMap.provenance]); return { - model, - atlas, - findings, - modelProvenance, + /** + * The publication seam (#335). In the product build `useDeliveredModel` is + * the identity function — no state, no effects, no behavioural difference. + * The benchmark application build defines the compile-time flag, which is + * where the artificial presentation-delay control withholds a newly accepted + * publication from the render tree. One publication (model + atlas + + * findings + provenance) is delayed as a unit so the render tree can never + * observe a split state. + */ + ...useDeliveredModel({ model, atlas, findings, modelProvenance }, model?.modelRevision ?? null), loading: snapshot.loading || runtimeMap.loading, error: snapshot.error ?? runtimeMap.error }; diff --git a/apps/web/src/lib/performance/modelAcceptance.tsx b/apps/web/src/lib/performance/modelAcceptance.tsx new file mode 100644 index 00000000..af9f7c50 --- /dev/null +++ b/apps/web/src/lib/performance/modelAcceptance.tsx @@ -0,0 +1,197 @@ +/** + * Benchmark-only instrumentation seam for the time-to-answer baseline (#335). + * + * This is REAL production application code, not a copy: `useSystemModel` calls + * `recordModelAcceptance()` at the exact point where a freshly fetched + * resource/revision pair becomes the coherent model the UI renders, and routes + * that publication through `useDeliveredModel()`. Both are gated by the + * compile-time constant `__DOCKERMAP_BENCH_ACCEPTANCE__`: + * + * - production (`apps/web/vite.config.ts`) defines it `false`, so the whole + * module collapses to `return value` / `return null` and every benchmark + * branch, event identifier and delay mechanism is eliminated from the shipped + * bundle (`tests/perf/productionIsolation.test.mjs` inspects the artifact); + * - the benchmark-mode application build in `tests/perf` defines it `true`, + * which is how the capture observes the real acceptance seam and how the + * stage-6/7 independence control injects an artificial presentation delay + * *after* acceptance. + * + * The seam carries no product payload and no telemetry. It records an opaque + * timing event plus the model revision token that was accepted, in an in-memory + * sink on the page, and (benchmark build only) stamps that same opaque token on + * the document root so the capture can prove the DOM content it times belongs to + * the accepted revision's render. Nothing leaves the page; nothing is uploaded. + */ +import { useEffect, useLayoutEffect, useRef, useState, type ReactElement } from "react"; +import type { DockerSnapshot, RuntimeMap } from "@dockermap/contracts"; +import { summarize, type SystemModel } from "../model"; + +/** One accepted coherent model: opaque timings only. */ +export interface ModelAcceptanceEvent { + /** Monotonic sequence number within this page. */ + seq: number; + /** `performance.now()` at the instant the coherent model was accepted. */ + at: number; + /** The opaque daemon model revision token the accepted model belongs to. */ + revision: string; + /** The exact snapshot/runtime-map pair consumed by useSystemModel. */ + snapshotRevision: string; + runtimeMapRevision: string; +} + +/** Benchmark-only diagnostic payload; it is drained by the capture harness. */ +export interface ModelLayerDiagnostic { + snapshot_revision: string; + runtime_map_revision: string; + snapshot_offline_count: number; + runtime_map_relevant_state: { revision: string; offline_or_not_running_service_count: number; offline_or_not_running_container_count: number }; + coherent_pair_accepted: { accepted: boolean; snapshot_revision: string; runtime_map_revision: string }; + derived_model_offline_value: number; + story_offline_value_pre_render: number; + rendered_home_offline_value: number | null; + fixture_generation: number | null; + monotonic_timestamp: number; +} + +declare global { +interface Window { + /** Benchmark build only. Absent from the production bundle. */ + __dockermapBenchAcceptanceSink?: ModelAcceptanceEvent[]; + /** Benchmark build only: artificial presentation delay in ms (0/absent = off). */ + __dockermapBenchRenderDelayMs?: number; +/** Revision-targeted delay used by the ordinary benchmark capture. */ +__dockermapBenchRenderDelayTarget?: string; +/** One-shot seam-isolation delay, armed before the next coherent acceptance. */ +__dockermapBenchDelayAfterNextAcceptance?: boolean; +/** Benchmark-only timestamp at which the post-acceptance delay timer began. */ +__dockermapBenchDelayStartedAt?: number; + /** Benchmark build only. Drained synchronously by the capture harness. */ + __dockermapBenchLayerSink?: ModelLayerDiagnostic[]; + /** Benchmark build only. Set by the harness before it advances a fixture. */ + __dockermapBenchFixtureGeneration?: number; + } +} + +const sink: ModelAcceptanceEvent[] = []; +const layerSink: ModelLayerDiagnostic[] = []; +let sequence = 0; +let lastAcceptedPair: string | null = null; + +/** + * The acceptance seam. Called from the real model publication path in + * `useSystemModel` at the moment `buildModel()` output becomes the model the UI + * uses — NOT from a DOM mutation, and NOT from a copied benchmark implementation. + * + * Duplicate calls for the same revision (a re-render recomputing the memo) are + * ignored, so one accepted revision produces exactly one event. + */ +export function recordModelAcceptance(snapshotRevision: string | null, runtimeMapRevision: string | null): void { + if (!__DOCKERMAP_BENCH_ACCEPTANCE__) return; + if (!snapshotRevision || snapshotRevision !== runtimeMapRevision) return; + const pair = `${snapshotRevision}\u0000${runtimeMapRevision}`; + if (pair === lastAcceptedPair) return; + lastAcceptedPair = pair; + sequence += 1; + sink.push({ seq: sequence, at: performance.now(), revision: snapshotRevision, snapshotRevision, runtimeMapRevision }); + // Bounded: the sink keeps only a recent window on a long-lived page. + if (sink.length > 256) sink.splice(0, 128); + window.__dockermapBenchAcceptanceSink = sink; +} + +/** Records the actual inputs and model value at the coherent-publication seam. */ +export function recordModelLayers(snapshot: DockerSnapshot, runtimeMap: RuntimeMap, model: SystemModel): void { + if (!__DOCKERMAP_BENCH_ACCEPTANCE__) return; + const runtimeStates = runtimeMap.nodes.filter((node) => /offline|stopped|dead|down|exited|not.running/i.test(`${node.status ?? ""} ${node.service?.status ?? ""}`)); + const diagnostic: ModelLayerDiagnostic = { + snapshot_revision: snapshot.modelRevision, + runtime_map_revision: runtimeMap.modelRevision, + snapshot_offline_count: snapshot.containers.filter((container) => /offline|stopped|dead|down|exited|not.running/i.test(container.status)).length, + runtime_map_relevant_state: { + revision: runtimeMap.modelRevision, + offline_or_not_running_service_count: runtimeStates.filter((node) => node.service !== undefined && node.service !== null).length, + offline_or_not_running_container_count: runtimeStates.filter((node) => node.type === "container").length + }, + coherent_pair_accepted: { accepted: snapshot.modelRevision === runtimeMap.modelRevision, snapshot_revision: snapshot.modelRevision, runtime_map_revision: runtimeMap.modelRevision }, + derived_model_offline_value: summarize(model).offline, + story_offline_value_pre_render: summarize(model).offline, + rendered_home_offline_value: null, + fixture_generation: window.__dockermapBenchFixtureGeneration ?? null, + monotonic_timestamp: performance.now() + }; + layerSink.push(diagnostic); + if (layerSink.length > 256) layerSink.splice(0, 128); + window.__dockermapBenchLayerSink = layerSink; +} + +/** + * The publication seam. In the product build this is the identity function: no + * state, no effects, no observable difference. In the benchmark build it is the + * point where the artificial presentation delay control withholds a newly + * accepted publication from the render tree. + * + * The delay is keyed on the accepted revision, never on the surrounding + * publication object (which is recreated on every render) — keying on the object + * would reschedule the timer on every render instead of once per publication. + */ +export function useDeliveredModel(value: T, revision: string | null): T { +if (!__DOCKERMAP_BENCH_ACCEPTANCE__) return value; +return useDelayedPublication(value, revision, window.__dockermapBenchRenderDelayTarget ?? null, window.__dockermapBenchDelayAfterNextAcceptance === true); +} + +/** + * Benchmark-only: stamps the accepted revision token on the document root in the + * same commit that renders the accepted model, so the capture can attribute a + * DOM repaint to a revision instead of assuming it. + */ +export function ModelAcceptanceStamp({ revision }: { revision: string | null }): ReactElement | null { + if (!__DOCKERMAP_BENCH_ACCEPTANCE__) return null; + return ; +} + +function AcceptedRevisionStamp({ revision }: { revision: string | null }): null { +useLayoutEffect(() => { + const root = document.documentElement; + if (revision && revision.length > 0) root.dataset.dockermapAcceptedRevision = revision; +else delete root.dataset.dockermapAcceptedRevision; + // This runs in the commit containing Home. Read its displayed metric instead + // of deriving a value from the revision or fixture generation. + const metric = [...document.querySelectorAll(".metric")].find((element) => + element.querySelector(".metric-label")?.textContent?.trim() === "Offline" + ); + const displayed = metric?.querySelector(".metric-value")?.textContent?.trim(); + const latest = layerSink[layerSink.length - 1]; + if (latest && latest.snapshot_revision === revision && displayed !== undefined && displayed !== "") { + const parsed = Number(displayed); + latest.rendered_home_offline_value = Number.isFinite(parsed) ? parsed : null; + } +}, [revision]); + return null; +} + +function useDelayedPublication(value: T, revision: string | null, delayTarget: string | null, delayAfterNextAcceptance: boolean): T { +const deliveredRevision = useRef(revision); +const isNextAcceptance = delayAfterNextAcceptance && Boolean(revision) && revision !== deliveredRevision.current; +const delayMs = revision === delayTarget || isNextAcceptance ? armedDelayMs() : 0; +const latest = useRef(value); +latest.current = value; +const [delivered, setDelivered] = useState(value); +useEffect(() => { +if (delayMs <= 0) return undefined; +// This flag is consumed only after recordModelAcceptance() ran in the same +// render. It deliberately identifies no daemon publication or trigger. +if (isNextAcceptance) window.__dockermapBenchDelayAfterNextAcceptance = false; +if (isNextAcceptance) window.__dockermapBenchDelayStartedAt = performance.now(); +const timer = window.setTimeout(() => { +deliveredRevision.current = revision; +setDelivered(latest.current); +}, delayMs); +return () => window.clearTimeout(timer); +}, [revision, delayMs, isNextAcceptance]); +return delayMs > 0 ? delivered : value; +} + +function armedDelayMs(): number { + if (typeof window === "undefined") return 0; + const raw = Number(window.__dockermapBenchRenderDelayMs ?? 0); + return Number.isFinite(raw) && raw > 0 ? raw : 0; +} diff --git a/apps/web/src/lib/performance/timeToAnswerEvidence.test.ts b/apps/web/src/lib/performance/timeToAnswerEvidence.test.ts new file mode 100644 index 00000000..429af697 --- /dev/null +++ b/apps/web/src/lib/performance/timeToAnswerEvidence.test.ts @@ -0,0 +1,275 @@ +import { describe, expect, it } from "vitest"; +import { + TIME_TO_ANSWER_BASELINE, + TIME_TO_ANSWER_CONTROLLED_RUNS, +TIME_TO_ANSWER_MATRIX, + TIME_TO_ANSWER_WARM_UP_METRICS, + TIME_TO_ANSWER_REFERENCE_FIXTURES, + TIME_TO_ANSWER_STAGES, + TIME_TO_ANSWER_WARMED_SAMPLES, + assertTimeToAnswerPromotion, + compatibleTimeToAnswerEnvironment, + derivedTimeToAnswerSummaries, + summarizeTimeToAnswerStage, + timeToAnswerLimit, + timeToAnswerP95, + validateTimeToAnswerEvidence, + withinTimeToAnswerPromotionLimit +} from "./timeToAnswerEvidence"; + +const environment: Record = { + runnerClass: "hearth-dedicated-x64", + cpuClass: "pinned-4-vcpu", + osImage: "ubuntu-24.04@sha256:fixture", + osKernel: "7.0.0-31-generic", + nodeRevision: "22.23.2", + rustRevision: "1.88.0", + dockerRevision: "29.0.0", + ssePollIntervalMs: "2000", + daemonBinarySha256: "eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee", + daemonBinaryBuild: "cargo-build-release-locked-p-dockermap-daemon", + cargoRevision: "cargo-1.88.0", + harnessRevision: "dddddddddddddddddddddddddddddddddddddddd", + browserEngine: "chromium", + browserRevision: "1234567", + browserFlags: ["--disable-background-networking"], + fontEnvironment: "Noto-Sans-1.0", + buildMode: "production", + fixtureRevision: "dockermap-v1/time-to-answer-fixtures-1", + sourceRevision: "candidate", +methodologyVersion: "dockermap-v1/time-to-answer-methodology-4" +}; + +/** + * Deliberately loosely typed: every hostile case below mutates the raw JSON a + * benchmark job would emit, and the validator must reject it without the test + * needing a cast per mutation. + */ +function rawEvidence(): { + baseline: string; + environment: Record; +records: { fixture: string; stage: string; measurementProtocol: string; sourceEvidenceFile: string; checkpointSha: string; runs: number[][] }[]; +} { + return { + baseline: TIME_TO_ANSWER_BASELINE, + environment, + records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }, record) => ({ +fixture, +stage, + measurementProtocol: stage === "publicationToNodeObservationMs" ? "controlled-poll-phase" : "end-to-end", + sourceEvidenceFile: stage === "publicationToNodeObservationMs" ? "stage-five.raw.json" : "general.raw.json", +checkpointSha: "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", +runs: Array.from({ length: TIME_TO_ANSWER_CONTROLLED_RUNS }, (_, run) => + Array.from( + { length: TIME_TO_ANSWER_WARMED_SAMPLES }, + (_, sample) => record * 100 + run * 10 + sample + 1 + ) + ) + })) + }; +} + +describe("time-to-answer evidence contract", () => { + it("defines a closed stage matrix covering every acceptance bucket", () => { + expect(new Set(TIME_TO_ANSWER_STAGES.map((stage) => stage.id)).size).toBe( + TIME_TO_ANSWER_STAGES.length + ); + expect([...new Set(TIME_TO_ANSWER_STAGES.map((stage) => stage.bucket))].sort()).toEqual([ + "backend-collection", + "browser-model", + "rendering", + "search", + "transport-notification" + ]); + // The three reference sizes are measured for every stage. + for (const stage of TIME_TO_ANSWER_STAGES) { + for (const reference of ["reference-25", "reference-100", "reference-250"]) { + expect(stage.fixtures).toContain(reference); + } + } + // Every listed fixture is a declared fixture, and the matrix is the union + // of the per-stage lists with no duplicate pair. + const declared = new Set(TIME_TO_ANSWER_REFERENCE_FIXTURES.map((fixture) => fixture.name)); + const expectedPairs = TIME_TO_ANSWER_STAGES.flatMap((stage) => stage.fixtures).length; + expect(TIME_TO_ANSWER_MATRIX.length).toBe(expectedPairs); + expect(new Set(TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => `${fixture}\u0000${stage}`)).size).toBe( + TIME_TO_ANSWER_MATRIX.length + ); + for (const { fixture } of TIME_TO_ANSWER_MATRIX) expect(declared.has(fixture)).toBe(true); + // The four scenario fixtures exist and are actually exercised. + for (const scenario of [ + "provider-only-revision-change", + "docker-topology-change", + "slow-bounded-compose-projection", + "unavailable-optional-provider" + ]) { + expect(declared.has(scenario)).toBe(true); + expect(TIME_TO_ANSWER_MATRIX.some(({ fixture }) => fixture === scenario)).toBe(true); + } + }); + +it("documents what each stage proves and does not prove", () => { + for (const stage of TIME_TO_ANSWER_STAGES) { + expect(stage.measures.length).toBeGreaterThan(20); + expect(stage.doesNotProve.length).toBeGreaterThan(20); + } + }); + + it("recomputes summaries from raw samples instead of trusting supplied values", () => { + const evidence = validateTimeToAnswerEvidence(rawEvidence()); + const summaries = derivedTimeToAnswerSummaries(evidence); + expect(TIME_TO_ANSWER_CONTROLLED_RUNS).toBe(3); + expect(TIME_TO_ANSWER_WARMED_SAMPLES).toBe(15); + expect(timeToAnswerP95(Array.from({ length: 15 }, (_, index) => index))).toBe(14); + const warmed = Array.from({ length: 15 }, (_, index) => index + 1); + expect(summarizeTimeToAnswerStage([warmed, warmed, warmed])).toEqual({ + runP95Ms: [15, 15, 15], + medianOfThreeRunP95Ms: 15, + reviewedMs: 15, + reviewedAggregation: "median-of-three-run-p95" + }); + const first = TIME_TO_ANSWER_MATRIX[0]!; + expect(summaries.get(`${first.fixture}\u0000${first.stage}`)).toEqual({ + runP95Ms: [15, 25, 35], + medianOfThreeRunP95Ms: 25, + reviewedMs: 25, + reviewedAggregation: "median-of-three-run-p95" + }); + }); + + it("fails closed on fabricated, incomplete or hostile artifacts", () => { + expect(() => + validateTimeToAnswerEvidence({ ...rawEvidence(), summary: "fabricated" }) + ).toThrow("closed baseline/environment/records schema"); + expect(() => + validateTimeToAnswerEvidence({ + ...rawEvidence(), + baseline: "dockermap-v1/other-baseline" + }) + ).toThrow("closed baseline/environment/records schema"); + expect(() => + validateTimeToAnswerEvidence({ + ...rawEvidence(), + records: rawEvidence().records.slice(1) + }) + ).toThrow("exact fixture × stage matrix"); + + const shortRun = rawEvidence(); + shortRun.records[0]!.runs[0] = [1]; + expect(() => validateTimeToAnswerEvidence(shortRun)).toThrow("exactly 15"); + + const twoRuns = rawEvidence(); + twoRuns.records[0]!.runs = twoRuns.records[0]!.runs.slice(0, 2); + expect(() => validateTimeToAnswerEvidence(twoRuns)).toThrow("three raw runs"); + + const negative = rawEvidence(); + negative.records[0]!.runs[0] = Array.from({ length: 15 }, () => -1); + expect(() => validateTimeToAnswerEvidence(negative)).toThrow("finite non-negative"); + + const notANumber = rawEvidence(); + notANumber.records[0]!.runs[0] = Array.from({ length: 15 }, () => "fast" as unknown as number); + expect(() => validateTimeToAnswerEvidence(notANumber)).toThrow("numeric"); + + const unknownStage = rawEvidence(); + unknownStage.records[0]!.stage = "vibesMs" as never; + expect(() => validateTimeToAnswerEvidence(unknownStage)).toThrow("unsafe or incomplete shape"); + +const duplicated = rawEvidence(); +duplicated.records[1] = { ...duplicated.records[0]! }; +expect(() => validateTimeToAnswerEvidence(duplicated)).toThrow("duplicate or unsupported"); + +const wrongProtocol = rawEvidence(); +wrongProtocol.records.find((record) => record.stage === "publicationToNodeObservationMs")!.measurementProtocol = "end-to-end"; + expect(() => validateTimeToAnswerEvidence(wrongProtocol)).toThrow("wrong measurement protocol"); + + const stageSixControl = rawEvidence(); + stageSixControl.records.find((record) => record.stage === "notificationToCoherentModelMs")!.measurementProtocol = "controlled-stage6-stage7-seam-isolation"; + expect(() => validateTimeToAnswerEvidence(stageSixControl)).toThrow("wrong measurement protocol"); + +const missingProvenance = rawEvidence(); +delete (missingProvenance.records[0] as Partial<(typeof missingProvenance.records)[number]>).checkpointSha; +expect(() => validateTimeToAnswerEvidence(missingProvenance)).toThrow("unsafe or incomplete shape"); + +const invalidCheckpoint = rawEvidence(); +invalidCheckpoint.records[0]!.checkpointSha = "not-a-checkpoint"; +expect(() => validateTimeToAnswerEvidence(invalidCheckpoint)).toThrow("provenance"); + + const undeclaredFixture = rawEvidence(); + undeclaredFixture.records[0]!.fixture = "reference-1000"; + expect(() => validateTimeToAnswerEvidence(undeclaredFixture)).toThrow( + "duplicate or unsupported" + ); + }); + + it("rejects unsafe or arbitrary runner metadata and only compares equivalent pinned environments", () => { + expect(() => + validateTimeToAnswerEvidence({ + ...rawEvidence(), + environment: { ...environment, rawHostPath: "/private/host" } + }) + ).toThrow("closed safe metadata fields"); + expect(() => + validateTimeToAnswerEvidence({ + ...rawEvidence(), + environment: { ...environment, fontEnvironment: "font with spaces" } + }) + ).toThrow("closed safe metadata fields"); + expect(() => + validateTimeToAnswerEvidence({ + ...rawEvidence(), + environment: { ...environment, browserEngine: "webkit" } + }) + ).toThrow("closed safe metadata fields"); + + const baseline = validateTimeToAnswerEvidence(rawEvidence()).environment; + expect(compatibleTimeToAnswerEnvironment(baseline, { ...baseline, sourceRevision: "other" })).toBe( + true + ); + expect(compatibleTimeToAnswerEnvironment(baseline, { ...baseline, browserRevision: "9" })).toBe( + false + ); + // dockerRevision is informational: no measured stage exercises the host + // Docker daemon, so a host engine upgrade must not invalidate a comparison. + expect(compatibleTimeToAnswerEnvironment(baseline, { ...baseline, dockerRevision: "30.0.0" })).toBe( + true + ); + // A dimension this benchmark really pins still has to match. + expect(compatibleTimeToAnswerEnvironment(baseline, { ...baseline, osImage: "debian-13" })).toBe( + false + ); + expect(compatibleTimeToAnswerEnvironment(baseline, { ...baseline, fixtureRevision: "v2" })).toBe( + false + ); + }); + + it("uses the reviewed max(baseline × 1.25, baseline + 2 ms) promotion gate", () => { + expect(timeToAnswerLimit(4)).toBe(6); + expect(timeToAnswerLimit(20)).toBe(25); + expect(withinTimeToAnswerPromotionLimit(20, 25)).toBe(true); + expect(withinTimeToAnswerPromotionLimit(20, 25.01)).toBe(false); + expect(() => timeToAnswerLimit(-1)).toThrow("finite non-negative"); + expect(() => timeToAnswerLimit(Number.NaN)).toThrow("finite non-negative"); + + const candidate = rawEvidence(); + candidate.environment = { ...candidate.environment, sourceRevision: "other" }; + expect(() => assertTimeToAnswerPromotion(rawEvidence(), candidate)).not.toThrow(); + + const slower = rawEvidence(); + slower.records[0]!.runs = slower.records[0]!.runs.map((run) => run.map(() => 1_000_000)); + expect(() => assertTimeToAnswerPromotion(rawEvidence(), slower)).toThrow( + "exceeds the reviewed promotion limit" + ); + + const wrongEnvironment = rawEvidence(); + wrongEnvironment.environment = { ...wrongEnvironment.environment, osImage: "debian-13" }; + expect(() => assertTimeToAnswerPromotion(rawEvidence(), wrongEnvironment)).toThrow( + "does not match the pinned baseline environment" + ); + }); +}); + + it("derives end-to-end calibration ownership from baseline protocol ownership", () => { + expect(TIME_TO_ANSWER_WARM_UP_METRICS).not.toContain("publicationToNodeObservationMs"); + expect(TIME_TO_ANSWER_WARM_UP_METRICS).toContain("notificationToCoherentModelMs"); + expect(TIME_TO_ANSWER_WARM_UP_METRICS).toContain("coherentModelToUsefulRenderMs"); + }); diff --git a/apps/web/src/lib/performance/timeToAnswerEvidence.ts b/apps/web/src/lib/performance/timeToAnswerEvidence.ts new file mode 100644 index 00000000..a1328789 --- /dev/null +++ b/apps/web/src/lib/performance/timeToAnswerEvidence.ts @@ -0,0 +1,981 @@ +import { phaseMediansMs, phaseNormalizedP95Ms } from "./timeToAnswerPollPhase"; + +/** + * DockerMap time-to-answer performance contract (#335). + * + * This module is the CLOSED schema and math for the controlled + * time-to-answer benchmark. It deliberately contains no timings and performs + * no measurement: the benchmark job runs on a pinned runner, writes a JSON + * artifact that stores RAW samples only, and this contract validates that + * artifact and derives every summary during review. + * + * Ordinary unit tests exercise this file's shape/math only. They must never + * compare elapsed time, be used as a performance gate, or be treated as + * evidence that DockerMap is fast. + */ + +/** + * Closed list of the stages the benchmark measures, in the order the operator + * path experiences them. `measures` is the number's meaning; `doesNotProve` + * is the part a reader must not infer from it. + */ +export const TIME_TO_ANSWER_STAGES = [ + { + id: "daemonStartToListenerMs", + bucket: "backend-collection", + measures: "Daemon process start until its HTTP listener accepts a request.", + doesNotProve: + "Nothing about collection. A fast listener with a slow first answer is still a slow product.", + fixtures: ["reference-25", "reference-100", "reference-250"] + }, + { + id: "listenerToFirstDockerModelMs", + bucket: "backend-collection", + measures: + "Listener readiness until the first authoritative Docker model is observable to a reader.", + doesNotProve: + "Does not include anything the browser does, and does not prove the model is complete: optional provider evidence may still be absent.", + fixtures: ["reference-25", "reference-100", "reference-250"] + }, + { + id: "dockerObservationMs", + bucket: "backend-collection", + measures: "One Docker inventory observation pass against the pinned fixture host.", + doesNotProve: + "Not a claim about a real Docker daemon's latency, host load, or image size; the fixture daemon is deterministic and local.", + fixtures: ["reference-25", "reference-100", "reference-250"] + }, + { + id: "composeEnrichmentMs", + bucket: "backend-collection", + measures: "Compose filesystem correlation for one publication, timed separately from the Docker observation.", + doesNotProve: + "Not a claim about a real Compose project tree; and while the stages are still coupled this number is measured, not removed (see #336).", + fixtures: ["reference-25", "reference-100", "reference-250", "slow-bounded-compose-projection"] + }, + { + id: "publicationToNodeObservationMs", + bucket: "transport-notification", + measures: + "Daemon publication until the Node/SSE layer observes that revision through TODAY'S real polling mechanism, poll wait included. The publication phase within the poll interval is DRIVEN, not hoped for: each recorded sample is assigned a declared phase on an explicit grid spanning the interval, the harness places the publication at that phase relative to its observation stream's poll ticks, and the observed phase is verified against the declared one before the sample is accepted.", + doesNotProve: + "Not browser work, not render, and not a claim about network distance to a remote operator. It is a phase response, not an observed user-traffic distribution: the reported phase-normalized summary weights the declared phases uniformly to characterise the latency the fixed polling mechanism imposes, and it does NOT claim that real host publications occur uniformly across poll phase.", + fixtures: [ + "reference-25", + "reference-100", + "reference-250", + "provider-only-revision-change", + "docker-topology-change", + "unavailable-optional-provider" + ] + }, + { + id: "notificationToCoherentModelMs", + bucket: "browser-model", + measures: + "Browser notification until the REAL application seam accepts one coherent model: the instant the fetched snapshot/runtime pair becomes the model the UI renders. It is observed at the application's own acceptance point, never derived from a DOM mutation.", + doesNotProve: + "Not a health judgement and not a statement that every evidence domain is current: it ends when a coherent model is accepted, not when the model is complete. It contains no rendering and says nothing about whether the operator saw anything.", + fixtures: [ + "reference-25", + "reference-100", + "reference-250", + "provider-only-revision-change", + "docker-topology-change", + "unavailable-optional-provider" + ] + }, + { + id: "coherentModelToUsefulRenderMs", + bucket: "rendering", + measures: + "From coherent-model acceptance until the accepted model's expected Home content is present — in a commit the application stamped with that accepted revision — followed by a bounded render/presentation confirmation (the probe observes the commit from an animation-frame loop and then awaits a bounded frame after it), and no sleeps. It shares no clock with the stage before it.", + doesNotProve: + "Not a visual-quality or accessibility claim, and not a claim that the operator found the answer. It is declared only for fixtures whose published change demonstrably repaints the Home content region; a provider-only or provider-unavailable revision is not guaranteed to repaint it, so measuring it there would be an empty number.", + fixtures: ["reference-25", "reference-100", "reference-250", "docker-topology-change"] + }, + { + id: "buildModelMs", + bucket: "browser-model", + measures: "One `buildModel()` composition for the fixture model.", + doesNotProve: "Nothing about rendering, network, or findings derivation.", + fixtures: ["reference-25", "reference-100", "reference-250"] + }, + { + id: "findingsDerivationMs", + bucket: "backend-collection", + measures: + "Findings derivation for the fixture's representative topology and evidence sizes. This runs in the daemon during publication, not in the browser.", + doesNotProve: + "Not a rule-quality claim, and it says nothing about a host with conditions the fixture does not contain. The fixture topology derives no findings, so this measures the empty-derivation path at its resolution floor.", + fixtures: ["reference-25", "reference-100", "reference-250"] + }, + { + id: "legacyTopologyLayoutMs", + bucket: "rendering", + measures: "The legacy Home topology layout (force-layout preview) for the fixture model.", + doesNotProve: + "Not a claim about Atlas, and it does not by itself justify removing the preview; #338 decides that from this evidence.", + fixtures: ["reference-25", "reference-100", "reference-250"] + }, + { + id: "commandQueryMs", + bucket: "search", + measures: "Cmd-K open plus query-to-results for the fixture's representative query classes.", + doesNotProve: + "Not a claim about answer quality, and the query set is a fixed representative sample, not operator behaviour.", + fixtures: ["reference-25", "reference-100", "reference-250"] + }, + { + id: "productionBundleMs", + bucket: "rendering", + measures: "Production bundle/loading cost for the pinned production build.", + doesNotProve: + "Not a transfer-time claim for a real network; it is measured against the pinned local build and recorded in the pinned environment.", + fixtures: ["reference-25", "reference-100", "reference-250"] + } +] as const; + +export type TimeToAnswerStage = (typeof TIME_TO_ANSWER_STAGES)[number]; +export type TimeToAnswerStageId = TimeToAnswerStage["id"]; +export type TimeToAnswerBucket = TimeToAnswerStage["bucket"]; +export type TimeToAnswerFixture = TimeToAnswerStage["fixtures"][number]; + +export const TIME_TO_ANSWER_BASELINE = "dockermap-v1/time-to-answer-baseline-4"; +export const TIME_TO_ANSWER_WARMED_SAMPLES = 15; +export const TIME_TO_ANSWER_CONTROLLED_RUNS = 3; + +/** + * The measurement design this contract describes. The baseline id names the + * CLOSED ARTIFACT SHAPE; the methodology version names HOW the numbers are + * produced — stage-5 deterministic phase control, the fixed burn-in policy, and + * the provenance/compatibility split. A candidate may + * only be compared against a baseline captured under the same methodology + * version, because a different design produces a different number for the same + * product. + */ +export const TIME_TO_ANSWER_METHODOLOGY = "dockermap-v1/time-to-answer-methodology-8"; + +/** + * Every ordinary warmed end-to-end run executes exactly 60 fixed conditioning + * observations followed by 15 measured observations. Observation 61 is always + * the first measured sample. The conditioning observations are retained for + * audit and never enter timing summaries or promotion comparisons. + * + * This is deliberately a workload contract, not a steady-state claim: it does + * not infer stationarity, adapt to values, or guarantee that a metric settles. + */ +export const TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS = 60; +/** + * Calibration is a separate, retained evidence exercise. Its constants are + * declared here (rather than in the runner) so a command cannot quietly tune + * them after it has seen an observation. + */ +// Fixed methodology-6 revision. Forty was selected after the previous +// protocol could not validate its own late derived count; it is not a +// statistically optimised or data-dependent window. +export const TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS = 60; +/** Historical calibration only; never Baseline-4 authority. */ +export const TIME_TO_ANSWER_WARM_UP_SAFETY_MARGIN = 2; + +/** + * Declared stationarity band: the median of the final two warm-up observations + * against the median of the measured window. This validity check is unchanged + * by the fixed-ten protocol revision. + */ +export const TIME_TO_ANSWER_STATIONARITY_MIN_RATIO = 0.5; +export const TIME_TO_ANSWER_STATIONARITY_MAX_RATIO = 1.5; + +/** + * Historical calibration is retained as a rejected, non-authoritative audit + * diagnostic. Its per-metric results cannot select, alter, or invalidate the + * Baseline-4 fixed 60-observation burn-in. + */ + +/** Reference fixtures (25/100/250 containers) plus the four scenario fixtures. */ +export const TIME_TO_ANSWER_REFERENCE_FIXTURES = [ + { name: "reference-25", containers: 25, kind: "reference" }, + { name: "reference-100", containers: 100, kind: "reference" }, + { name: "reference-250", containers: 250, kind: "reference" }, + { name: "provider-only-revision-change", containers: 100, kind: "scenario" }, + { name: "docker-topology-change", containers: 100, kind: "scenario" }, + { name: "slow-bounded-compose-projection", containers: 100, kind: "scenario" }, + { name: "unavailable-optional-provider", containers: 100, kind: "scenario" } +] as const; + +/** The closed fixture × stage matrix, derived from each stage's fixture list. */ +export const TIME_TO_ANSWER_MATRIX = TIME_TO_ANSWER_STAGES.flatMap((stage) => +stage.fixtures.map((fixture) => ({ fixture, stage: stage.id as TimeToAnswerStageId })) +); + +/** Baseline-cell ownership is the authority for timing evidence producers. */ +export type TimeToAnswerBaselineProtocol = "controlled-poll-phase" | "end-to-end"; +export function timeToAnswerBaselineProtocol(stage: TimeToAnswerStageId): TimeToAnswerBaselineProtocol { + return stage === "publicationToNodeObservationMs" ? "controlled-poll-phase" : "end-to-end"; +} +export const TIME_TO_ANSWER_END_TO_END_MATRIX = TIME_TO_ANSWER_MATRIX.filter( + (cell) => timeToAnswerBaselineProtocol(cell.stage) === "end-to-end" +); +export const TIME_TO_ANSWER_CONTROLLED_POLL_MATRIX = TIME_TO_ANSWER_MATRIX.filter( + (cell) => timeToAnswerBaselineProtocol(cell.stage) === "controlled-poll-phase" +); + +/** + * Every pinned dimension of a controlled run. A missing field, or a candidate + * whose pinned environment differs from the baseline, invalidates the record; + * it never justifies retrying until a preferred duration appears. + */ +export type TimeToAnswerEnvironment = { + runnerClass: string; + cpuClass: string; + osImage: string; + osKernel: string; + nodeRevision: string; + rustRevision: string; + dockerRevision: string; + /** + * The effective `DOCKERMAP_SSE_INTERVAL_MS` the API ran with. Stage 5 + * measures today's real publication-observation mechanism, poll wait + * included, so the interval is part of the pinned environment: a candidate + * that changed it has not been measured against the same mechanism. + */ + ssePollIntervalMs: string; + /** + * The revision of the benchmark harness itself (latest commit touching + * `tests/perf` and the performance contract). A baseline is only reproducible + * if both the product and the harness that measured it are identified: a + * number produced by an uncommitted harness cannot be re-derived by anyone. + */ + /** sha256 of the exact release daemon executable this capture ran. */ + daemonBinarySha256: string; + /** The command and profile that produced that binary. */ + daemonBinaryBuild: string; + cargoRevision: string; + harnessRevision: string; + browserEngine: "chromium"; + browserRevision: string; + browserFlags: readonly string[]; + fontEnvironment: string; + buildMode: "production"; + fixtureRevision: string; + sourceRevision: string; + /** The measurement design (see TIME_TO_ANSWER_METHODOLOGY). Required to match. */ + methodologyVersion: string; +}; + +export interface TimeToAnswerRecord { +fixture: string; +stage: TimeToAnswerStageId; +/** Controlled sub-benchmarks own only the cells they explicitly name. */ + measurementProtocol: "controlled-poll-phase" | "controlled-stage6-stage7-seam-isolation" | "end-to-end"; +/** Raw evidence section that produced this one cell. */ +sourceEvidenceFile: string; +/** Committed source/harness checkpoint that produced this cell. */ +checkpointSha: string; +/** Each inner array is one complete controlled run of warmed samples. */ +runs: readonly (readonly number[])[]; +} + +export interface TimeToAnswerEvidence { + baseline: typeof TIME_TO_ANSWER_BASELINE; + environment: TimeToAnswerEnvironment; + records: readonly TimeToAnswerRecord[]; +} + +export interface TimeToAnswerStageSummary { + runP95Ms: readonly number[]; + medianOfThreeRunP95Ms: number; + /** The authority used for review and promotion of this record. */ + reviewedMs: number; + reviewedAggregation: "median-of-three-run-p95" | "phase-normalized-p95"; +} + +const environmentKeys = [ + "runnerClass", + "cpuClass", + "osImage", + "osKernel", + "nodeRevision", + "rustRevision", + "dockerRevision", + "ssePollIntervalMs", + "harnessRevision", + "daemonBinarySha256", + "daemonBinaryBuild", + "cargoRevision", + "browserEngine", + "browserRevision", + "browserFlags", + "fontEnvironment", + "buildMode", + "fixtureRevision", + "sourceRevision", + "methodologyVersion" +] as const; +const evidenceKeys = ["baseline", "environment", "records"] as const; +const recordKeys = ["fixture", "stage", "measurementProtocol", "sourceEvidenceFile", "checkpointSha", "runs"] as const; +const safeValue = /^[A-Za-z0-9._/@:+=-]{1,160}$/; +const checkpointSha = /^[0-9a-f]{7,40}$/; +const stageIds = new Set(TIME_TO_ANSWER_STAGES.map((stage) => stage.id)); + +function isObject(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +function hasExactKeys(value: Record, keys: readonly string[]): boolean { + const actual = Object.keys(value).sort(); + const expected = [...keys].sort(); + return actual.length === expected.length && actual.every((key, index) => key === expected[index]); +} + +function safeString(value: unknown): value is string { + return typeof value === "string" && safeValue.test(value); +} + +/** Nearest-rank percentile: for 15 samples, p95 is the largest observed value. */ +export function timeToAnswerP95(samples: readonly number[]): number { + if ( + samples.length !== TIME_TO_ANSWER_WARMED_SAMPLES || + samples.some((sample) => typeof sample !== "number" || !Number.isFinite(sample) || sample < 0) + ) { + throw new Error( + `Time-to-answer benchmark requires exactly ${TIME_TO_ANSWER_WARMED_SAMPLES} finite non-negative warmed samples.` + ); + } + const sorted = [...samples].sort((left, right) => left - right); + return sorted[Math.ceil(sorted.length * 0.95) - 1]!; +} + +export function summarizeTimeToAnswerStage( + runs: readonly (readonly number[])[] +): TimeToAnswerStageSummary { + if (runs.length !== TIME_TO_ANSWER_CONTROLLED_RUNS) { + throw new Error( + `Time-to-answer benchmark requires exactly ${TIME_TO_ANSWER_CONTROLLED_RUNS} complete controlled runs.` + ); + } + const runP95Ms = runs.map(timeToAnswerP95); + const ordered = [...runP95Ms].sort((left, right) => left - right); + return { + runP95Ms, + medianOfThreeRunP95Ms: ordered[1]!, + reviewedMs: ordered[1]!, + reviewedAggregation: "median-of-three-run-p95" + }; +} + +export function assertTimeToAnswerEnvironment( + environment: unknown +): asserts environment is TimeToAnswerEnvironment { + if ( + !isObject(environment) || + !hasExactKeys(environment, environmentKeys) || + environment.browserEngine !== "chromium" || + environment.buildMode !== "production" || + ![ + environment.runnerClass, + environment.cpuClass, + environment.osImage, + environment.osKernel, + environment.nodeRevision, + environment.rustRevision, + environment.dockerRevision, + environment.ssePollIntervalMs, + environment.harnessRevision, + environment.daemonBinarySha256, + environment.daemonBinaryBuild, + environment.cargoRevision, + environment.browserRevision, + environment.fontEnvironment, + environment.fixtureRevision, + environment.sourceRevision, + environment.methodologyVersion + ].every(safeString) || + !Array.isArray(environment.browserFlags) || + environment.browserFlags.length === 0 || + environment.browserFlags.length > 16 || + !environment.browserFlags.every(safeString) + ) { + throw new Error( + "Time-to-answer benchmark environment must use exactly the closed safe metadata fields for pinned runner/CPU/OS/kernel, Node/Rust/Docker revisions, Chromium revision and flags, fonts, production build, and fixture/source revision." + ); + } +} + +/** + * Reject untrusted JSON before deriving summaries. The artifact stores raw + * samples only; a supplied summary is not accepted as input. + */ +export function validateTimeToAnswerEvidence(value: unknown): TimeToAnswerEvidence { + if ( + !isObject(value) || + !hasExactKeys(value, evidenceKeys) || + value.baseline !== TIME_TO_ANSWER_BASELINE || + !Array.isArray(value.records) + ) { + throw new Error("Time-to-answer evidence must use the closed baseline/environment/records schema."); + } + assertTimeToAnswerEnvironment(value.environment); + const expected = new Set(TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => `${fixture}\u0000${stage}`)); + if (value.records.length !== expected.size) { + throw new Error("Time-to-answer evidence must contain the exact fixture × stage matrix."); + } + const records = value.records.map((raw) => { + if ( +!isObject(raw) || +!hasExactKeys(raw, recordKeys) || +typeof raw.fixture !== "string" || +typeof raw.stage !== "string" || +!stageIds.has(raw.stage) || +typeof raw.measurementProtocol !== "string" || +typeof raw.sourceEvidenceFile !== "string" || +typeof raw.checkpointSha !== "string" || +!Array.isArray(raw.runs) + ) { + throw new Error("Time-to-answer record has an unsafe or incomplete shape."); + } + const key = `${raw.fixture}\u0000${raw.stage}`; +if (!expected.delete(key)) { +throw new Error("Time-to-answer evidence has a duplicate or unsupported fixture/stage record."); +} + const requiredProtocol = timeToAnswerBaselineProtocol(raw.stage as TimeToAnswerStageId); +if (raw.measurementProtocol !== requiredProtocol) { +throw new Error(`Time-to-answer record ${raw.fixture}/${raw.stage} has the wrong measurement protocol.`); +} +if (!safeString(raw.sourceEvidenceFile) || !checkpointSha.test(raw.checkpointSha)) { +throw new Error("Time-to-answer record provenance must contain safe source evidence and checkpoint identifiers."); +} + if ( + raw.runs.length !== TIME_TO_ANSWER_CONTROLLED_RUNS || + !raw.runs.every((run) => Array.isArray(run)) + ) { + throw new Error("Time-to-answer evidence requires exactly three raw runs per stage."); + } + const runs = raw.runs.map((run) => + (run as unknown[]).map((sample) => { + if (typeof sample !== "number") throw new Error("Time-to-answer samples must be numeric."); + return sample; + }) + ); + // Executes the finite/non-negative/sample-count checks so summaries cannot + // be trusted input, and so a fabricated summary field cannot survive. + summarizeTimeToAnswerStage(runs); +return { +fixture: raw.fixture, +stage: raw.stage as TimeToAnswerStageId, +measurementProtocol: raw.measurementProtocol as TimeToAnswerRecord["measurementProtocol"], +sourceEvidenceFile: raw.sourceEvidenceFile, +checkpointSha: raw.checkpointSha, +runs +}; + }); + if (expected.size !== 0) { + throw new Error("Time-to-answer evidence is missing a required fixture/stage record."); + } + return { baseline: TIME_TO_ANSWER_BASELINE, environment: value.environment, records }; +} + +export function derivedTimeToAnswerSummaries( + evidence: TimeToAnswerEvidence +): ReadonlyMap { + return new Map( + evidence.records.map((record) => { + const summary = summarizeTimeToAnswerStage(record.runs); +if (record.measurementProtocol === "controlled-poll-phase") { + const normalized = derivedTimeToAnswerPhaseNormalized(record.runs, evidence.environment.ssePollIntervalMs); + return [ + `${record.fixture}\u0000${record.stage}`, + { + ...summary, + reviewedMs: normalized.phaseNormalizedP95Ms, + reviewedAggregation: "phase-normalized-p95" as const + } + ]; + } + return [`${record.fixture}\u0000${record.stage}`, summary]; + }) + ); +} + +/** Source revision deliberately differs between a baseline and its candidate. */ +/** + * What kind of measurement each stage is. The distinction is load-bearing, not + * descriptive: a warmed stage is a repeated steady-state operation whose first + * observation is a cold start, so that observation is recorded separately as + * warm-up and never enters the summary. A cold-start stage is the opposite — + * its first observation IS the measurement. Scenario-specific stages are only + * declared for fixtures that deliberately construct the scenario. + */ +export const TIME_TO_ANSWER_STAGE_KIND: Record = { + daemonStartToListenerMs: "cold-start", + listenerToFirstDockerModelMs: "cold-start", + dockerObservationMs: "warmed-repeated", + composeEnrichmentMs: "warmed-repeated", + publicationToNodeObservationMs: "warmed-repeated", + notificationToCoherentModelMs: "warmed-repeated", + coherentModelToUsefulRenderMs: "warmed-repeated", + buildModelMs: "warmed-repeated", + findingsDerivationMs: "warmed-repeated", + legacyTopologyLayoutMs: "warmed-repeated", + commandQueryMs: "warmed-repeated", + productionBundleMs: "warmed-repeated" +}; + +/** A stage measured on a scenario fixture is scenario-specific for that cell. */ +export function isScenarioCell(fixture: string, stage: string): boolean { + const declared = TIME_TO_ANSWER_REFERENCE_FIXTURES.find((entry) => entry.name === fixture); + return declared?.kind === "scenario" && TIME_TO_ANSWER_STAGE_KIND[stage] === "warmed-repeated"; +} + +export type WarmUpCalibrationCell = { + fixture: string; + metric: string; + observations: readonly number[]; +}; + +export type WarmUpCalibrationDerivation = { + metric: string; + fixtureCounts: Readonly>; + frozenWarmUpCount: number; +}; + +export type WarmUpCalibrationCandidate = { candidate: number; ratio: number | null; inBand: boolean; sustained: boolean }; +export type WarmUpCalibrationFixtureReport = { fixture: string; stableWarmUpCount: number | null; candidates: readonly WarmUpCalibrationCandidate[]; reason: string | null }; +export type WarmUpCalibrationMetricReport = { + metric: string; + fixtures: readonly WarmUpCalibrationFixtureReport[]; + maximumStableWarmUpCount: number | null; + proposedWarmUpCount: number | null; + requiredEvidenceLength: number | null; + evidenceBacked: boolean; + verdict: "PASS" | "CONFLICT"; + reason: string | null; +}; +export type WarmUpCalibrationReport = { + verdict: "PASS" | "CONFLICT"; + metrics: readonly WarmUpCalibrationMetricReport[]; + /** Historical diagnostic only; never Baseline-4 authority. */ + nonAuthoritativeProposedWarmUpCounts: Readonly>; +}; + +/** + * The calibration population is closed independently of the baseline matrix: + * only the three size reference fixtures determine a warmed metric's count. + * Scenario cells are deliberately not a source of conditioning evidence. + */ +export const TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES = TIME_TO_ANSWER_REFERENCE_FIXTURES + .filter((fixture) => fixture.kind === "reference") + .map((fixture) => fixture.name); + +/** Every repeated end-to-end stage must be calibrated before baseline capture. */ +export const TIME_TO_ANSWER_WARM_UP_METRICS = TIME_TO_ANSWER_STAGES +.filter((stage) => + TIME_TO_ANSWER_STAGE_KIND[stage.id] === "warmed-repeated" && + TIME_TO_ANSWER_END_TO_END_MATRIX.some((cell) => cell.stage === stage.id) +) +.map((stage) => stage.id); + +/** Median used by the existing stationarity semantics. */ +function median(values: readonly number[]): number { + if (values.length === 0) return Number.NaN; + const ordered = [...values].sort((left, right) => left - right); + const middle = Math.floor(ordered.length / 2); + return ordered.length % 2 === 0 ? (ordered[middle - 1]! + ordered[middle]!) / 2 : ordered[middle]!; +} + +/** + * Derive one metric's frozen count from its complete fixed-window reference + * fixture cells. Candidate `w` compares obs[w-2:w] to obs[w:w+15]. The first + * candidate whose ratio stays inside the declared band at every later eligible + * position is selected for each fixture; the metric receives their maximum plus + * the fixed safety margin. The margin itself must still be evidence-backed. + */ +export function deriveFrozenWarmUpCount(cells: readonly WarmUpCalibrationCell[]): WarmUpCalibrationDerivation { + if (cells.length === 0) throw new Error("warm-up calibration needs at least one relevant reference fixture"); + const metric = cells[0]!.metric; + if (cells.some((cell) => cell.metric !== metric)) throw new Error("warm-up calibration derives one metric at a time"); + if (!TIME_TO_ANSWER_WARM_UP_METRICS.includes(metric as TimeToAnswerStageId)) { + throw new Error(`${metric} is not a warmed end-to-end metric`); + } + const expectedFixtures = new Set(TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES); + if (cells.length !== expectedFixtures.size) { + throw new Error(`${metric} calibration must retain every declared reference fixture`); + } + const fixtureCounts: Record = {}; + const latestEligible = TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS - TIME_TO_ANSWER_WARMED_SAMPLES; + for (const cell of cells) { + if (!cell.fixture || fixtureCounts[cell.fixture] !== undefined || !expectedFixtures.delete(cell.fixture)) { + throw new Error("warm-up calibration fixtures must be the unique declared reference fixtures"); + } + if (cell.observations.length !== TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS || cell.observations.some((value) => !Number.isFinite(value) || value < 0)) { + throw new Error(`${cell.fixture}/${metric} must retain exactly ${TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS} finite non-negative calibration observations`); + } + const stableAt = (candidate: number) => { + for (let position = candidate; position <= latestEligible; position += 1) { + const ratio = median(cell.observations.slice(position - 2, position)) / median(cell.observations.slice(position, position + TIME_TO_ANSWER_WARMED_SAMPLES)); + if (!Number.isFinite(ratio) || ratio <= 0 || ratio < TIME_TO_ANSWER_STATIONARITY_MIN_RATIO || ratio > TIME_TO_ANSWER_STATIONARITY_MAX_RATIO) return false; + } + return true; + }; + const earliest = Array.from({ length: latestEligible - 1 }, (_, index) => index + 2).find(stableAt); + if (earliest === undefined) throw new Error(`${cell.fixture}/${metric} never reaches sustained stationarity in the retained calibration window`); + fixtureCounts[cell.fixture] = earliest; + } + if (expectedFixtures.size !== 0) { + throw new Error(`${metric} calibration is missing declared reference fixtures: ${[...expectedFixtures].join(", ")}`); + } + const frozenWarmUpCount = Math.max(...Object.values(fixtureCounts)) + TIME_TO_ANSWER_WARM_UP_SAFETY_MARGIN; + if (frozenWarmUpCount > latestEligible) { + throw new Error(`${metric} calibration conflict: safety margin ${TIME_TO_ANSWER_WARM_UP_SAFETY_MARGIN} moves warm-up count ${frozenWarmUpCount} beyond the evidence-backed ${TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS}-observation window`); + } + return { metric, fixtureCounts, frozenWarmUpCount }; +} + +/** Derive every metric before returning an atomic overall calibration verdict. */ +export function deriveWarmUpCalibrationReport(cells: readonly WarmUpCalibrationCell[]): WarmUpCalibrationReport { + const latest = TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS - TIME_TO_ANSWER_WARMED_SAMPLES; + const metrics = TIME_TO_ANSWER_WARM_UP_METRICS.map((metric) => { + const metricCells = cells.filter((cell) => cell.metric === metric); + const fixtures = TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES.map((fixture) => { + const matches = metricCells.filter((cell) => cell.fixture === fixture); + const cell = matches.length === 1 ? matches[0] : undefined; + if (!cell || cell.observations.length !== TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS || cell.observations.some((value) => !Number.isFinite(value) || value < 0)) return { fixture, stableWarmUpCount: null, candidates: [], reason: !cell ? "missing reference fixture" : matches.length !== 1 ? "duplicate reference fixture" : `must retain exactly ${TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS} finite non-negative calibration observations` }; + const candidates = Array.from({ length: latest - 1 }, (_, index) => index + 2).map((candidate) => { + const ratio = median(cell.observations.slice(candidate - 2, candidate)) / median(cell.observations.slice(candidate, candidate + TIME_TO_ANSWER_WARMED_SAMPLES)); + return { candidate, ratio: Number.isFinite(ratio) ? ratio : null, inBand: Number.isFinite(ratio) && ratio > 0 && ratio >= TIME_TO_ANSWER_STATIONARITY_MIN_RATIO && ratio <= TIME_TO_ANSWER_STATIONARITY_MAX_RATIO, sustained: false }; + }); + const traced = candidates.map((candidate, index) => ({ ...candidate, sustained: candidate.inBand && candidates.slice(index).every((later) => later.inBand) })); + const stableWarmUpCount = traced.find((candidate) => candidate.sustained)?.candidate ?? null; + return { fixture, stableWarmUpCount, candidates: traced, reason: stableWarmUpCount === null ? "never reaches sustained stationarity in the retained calibration window" : null }; + }); + const expectedFixtures = new Set(TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES); + const unsupported = metricCells.some((cell) => !expectedFixtures.has(cell.fixture)); + const counts = fixtures.map((fixture) => fixture.stableWarmUpCount); + const maximumStableWarmUpCount = counts.every((count): count is number => count !== null) ? Math.max(...counts) : null; + const proposedWarmUpCount = maximumStableWarmUpCount === null ? null : maximumStableWarmUpCount + TIME_TO_ANSWER_WARM_UP_SAFETY_MARGIN; + const requiredEvidenceLength = proposedWarmUpCount === null ? null : proposedWarmUpCount + TIME_TO_ANSWER_WARMED_SAMPLES; + const evidenceBacked = requiredEvidenceLength !== null && requiredEvidenceLength <= TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS; + const reason = unsupported ? "contains unsupported reference fixture" : fixtures.find((fixture) => fixture.reason)?.reason ?? (!evidenceBacked ? `safety margin requires ${requiredEvidenceLength} observations, exceeding the fixed ${TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS}-observation window` : null); + return { metric, fixtures, maximumStableWarmUpCount, proposedWarmUpCount, requiredEvidenceLength, evidenceBacked, verdict: reason ? "CONFLICT" as const : "PASS" as const, reason }; + }); + const passed = metrics.every((metric) => metric.verdict === "PASS"); + return { verdict: passed ? "PASS" : "CONFLICT", metrics, nonAuthoritativeProposedWarmUpCounts: Object.freeze(passed ? Object.fromEntries(metrics.map((metric) => [metric.metric, metric.proposedWarmUpCount!])) : {}) }; +} + +/** + * Split one warmed measurement window into fixed burn-in + * observations and the recorded samples. + * + * The daemon's first-ever refresh runs before its listener binds, so its first + * passes through the collection path are cold. The count is FIXED by protocol + * (`TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS`), never chosen by looking at the data: + * with 15 recorded samples, nearest-rank p95 is the maximum, so a surviving cold + * observation would otherwise *become* the published number. Every warm-up + * observation is returned for the raw audit trail, and none of them enters the + * summary. + */ +export function splitWarmedObservations( +observations: readonly number[], +count = TIME_TO_ANSWER_WARMED_SAMPLES, + burnInCount = TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS +): { warmUps: number[]; recorded: number[] } { + const required = count + burnInCount; + if (observations.length < required) { + throw new Error( + `a warmed stage needs at least ${required} observations: ${burnInCount} fixed ` + + `burn-in observations plus ${count} recorded samples` + ); + } + const warmUps = observations.slice(0, burnInCount); + if ( + [...warmUps, ...observations.slice(0, required)].some( + (value) => typeof value !== "number" || !Number.isFinite(value) || value < 0 + ) + ) { + throw new Error("warm-up and recorded observations must be finite non-negative numbers"); + } + return { warmUps: [...warmUps], recorded: observations.slice(burnInCount, required) as number[] }; +} + +/** + * Declared stationarity check for one warmed cell/run. Compares the FINAL + * warm-up observations against the measured window using the predeclared band, + * and returns the ratio for the audit trail. A window whose warm-ups have not + * settled is INVALID — it is never repaired by discarding further samples, + * because choosing how many samples to drop after seeing the values would turn + * benchmark conditioning into result selection. + */ +export function assertWarmUpStationarity(input: { + label: string; +warmUps: readonly number[]; +recorded: readonly number[]; + warmUpCount: number; +}): number { + const { label, warmUps, recorded, warmUpCount } = input; + if (warmUps.length !== warmUpCount) { + throw new Error(`${label} must retain exactly ${warmUpCount} warm-up observations`); + } + if (recorded.length !== TIME_TO_ANSWER_WARMED_SAMPLES) { + throw new Error(`${label} must record exactly ${TIME_TO_ANSWER_WARMED_SAMPLES} measured samples`); + } + const ratio = warmUpStationarityCalculation(warmUps, recorded).ratio; + if (!Number.isFinite(ratio) || ratio <= 0) { + throw new Error(`${label} has no usable warm-up/measured ratio`); + } + if (ratio > TIME_TO_ANSWER_STATIONARITY_MAX_RATIO || ratio < TIME_TO_ANSWER_STATIONARITY_MIN_RATIO) { + throw new Error( + `${label} is not stationary: the final warm-up observations sit at ${ratio.toFixed(2)}x the measured ` + + `median, outside the declared ${TIME_TO_ANSWER_STATIONARITY_MIN_RATIO}–` + + `${TIME_TO_ANSWER_STATIONARITY_MAX_RATIO}x band` + ); + } + return ratio; +} + +/** The audit calculation used by the unchanged warm-up stationarity validity check. */ +export function warmUpStationarityCalculation( + warmUps: readonly number[], + recorded: readonly number[] +): { finalWarmUpMedian: number; measuredMedian: number; ratio: number } { + const finalWarmUpMedian = median(warmUps.slice(-2)); + const measuredMedian = median(recorded); + return { finalWarmUpMedian, measuredMedian, ratio: finalWarmUpMedian / measuredMedian }; +} + +/** + * Bind the executed daemon binary to the recorded source revision. The benchmark + * does not claim bit-for-bit reproducible Rust builds across machines; it proves + * which binary THIS capture executed. + */ +export function assertDaemonBinaryProvenance(input: { + expectedSha256: string; + observedSha256: string; + phase: string; +}): void { + if (!/^[0-9a-f]{64}$/.test(input.expectedSha256) || !/^[0-9a-f]{64}$/.test(input.observedSha256)) { + throw new Error("daemon binary provenance requires two lowercase sha256 digests"); + } + if (input.expectedSha256 !== input.observedSha256) { + throw new Error( + `daemon binary provenance failed ${input.phase}: the executable is not the binary this capture pinned` + ); + } +} + +/** + * Provenance/identity keys: recorded so a baseline identifies exactly what was + * measured, but never a comparison REQUIREMENT. The executable digest and the + * source/harness revisions differ by construction for any legitimate candidate + * that changes the product or the harness, so requiring them to match would make + * comparison impossible — and the daemon binary is rebuilt from the candidate + * checkout, so a byte-identical digest is not even reproducible across a changed + * `CARGO_HOME`. + */ +export const TIME_TO_ANSWER_PROVENANCE_KEYS = [ + "sourceRevision", + "harnessRevision", + "daemonBinarySha256" +] as const; + +/** + * Recorded but informational: no measured stage exercises the host Docker daemon + * (the capture runs against the deterministic fixture daemon), so requiring this + * to match would fail a comparison for a dimension this benchmark never touches. + */ +export const TIME_TO_ANSWER_INFORMATIONAL_KEYS = ["dockerRevision"] as const; + +/** + * The keys a candidate must share with the baseline to be comparable at all: + * runner/CPU/OS/kernel, Node/Rust toolchain, cargo, browser engine/revision/ + * flags, fonts, production build mode, fixture revision, the polling + * configuration, the daemon build command, and the benchmark methodology + * version. A time-to-answer number is only comparable to another number produced + * by the same design in the same environment. + */ +export const timeToAnswerCompatibilityKeys = environmentKeys.filter( + (key) => + !(TIME_TO_ANSWER_PROVENANCE_KEYS as readonly string[]).includes(key) && + !(TIME_TO_ANSWER_INFORMATIONAL_KEYS as readonly string[]).includes(key) +); + +export function compatibleTimeToAnswerEnvironment( + baseline: TimeToAnswerEnvironment, + candidate: TimeToAnswerEnvironment +): boolean { + return timeToAnswerCompatibilityKeys.every( + (key) => JSON.stringify(baseline[key]) === JSON.stringify(candidate[key]) + ); +} + +/** + * The phase-normalized stage-5 figure (methodology revision 2): the observed + * latency median at each DECLARED phase, then nearest-rank p95 over those phase + * medians. Uniform weighting over the declared grid is a statement about the + * polling MECHANISM — never about real host publication phase or network + * distance. + */ +export function derivedTimeToAnswerPhaseNormalized( + runs: readonly (readonly number[])[], + ssePollIntervalMs: string +): { phaseMediansMs: readonly number[]; phaseNormalizedP95Ms: number } { + const intervalMs = Number(ssePollIntervalMs); + if (!Number.isFinite(intervalMs) || intervalMs <= 0) { + throw new Error("the pinned SSE poll interval must be a positive number of milliseconds"); + } + if (runs.length !== TIME_TO_ANSWER_CONTROLLED_RUNS) { + throw new Error(`stage 5 requires exactly ${TIME_TO_ANSWER_CONTROLLED_RUNS} controlled runs to normalise by phase`); + } + return { + phaseMediansMs: phaseMediansMs(runs, intervalMs), + phaseNormalizedP95Ms: phaseNormalizedP95Ms(runs, intervalMs) + }; +} + +/** + * Regression limits are derived from the measured baseline with the same + * reviewed rule the Atlas evidence uses; no aspirational absolute millisecond + * budget is invented before a baseline exists. + */ +export function timeToAnswerLimit(baselineMs: number): number { + if (!Number.isFinite(baselineMs) || baselineMs < 0) { + throw new Error("Time-to-answer baseline must be a finite non-negative duration."); + } + return Math.max(baselineMs * 1.25, baselineMs + 2); +} + +export function withinTimeToAnswerPromotionLimit(baselineMs: number, candidateMs: number): boolean { + return Number.isFinite(candidateMs) && candidateMs >= 0 && candidateMs <= timeToAnswerLimit(baselineMs); +} + +/** The benchmark job calls this after reading two closed JSON artifacts. */ +export function assertTimeToAnswerPromotion(baselineRaw: unknown, candidateRaw: unknown): void { + const baseline = validateTimeToAnswerEvidence(baselineRaw); + const candidate = validateTimeToAnswerEvidence(candidateRaw); + if (!compatibleTimeToAnswerEnvironment(baseline.environment, candidate.environment)) { + throw new Error("Time-to-answer candidate does not match the pinned baseline environment."); + } + const baselineSummaries = derivedTimeToAnswerSummaries(baseline); + for (const [key, candidateSummary] of derivedTimeToAnswerSummaries(candidate)) { + const baselineSummary = baselineSummaries.get(key); + if ( + !baselineSummary || + !withinTimeToAnswerPromotionLimit( + baselineSummary.reviewedMs, + candidateSummary.reviewedMs + ) + ) { + throw new Error(`Time-to-answer candidate exceeds the reviewed promotion limit for ${key}.`); + } + } +} + +/* ------------------------------------------------------------------ * + * Stage 6 / stage 7 independence control (#335) + * + * Stage 6 ends when the APPLICATION accepts a coherent model; stage 7 begins + * at that instant and ends when the accepted model's expected Home content has + * rendered (and one bounded frame has confirmed presentation). If the two + * numbers came from one clock, an artificial presentation delay injected AFTER + * acceptance would move both. The control therefore arms exactly that delay and + * requires stage 6 to stay put while stage 7 grows by the injected amount. + * ------------------------------------------------------------------ */ + +/** The artificial presentation delay injected after coherent-model acceptance. */ +export const TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS = 250; +/** Control samples per fixture that declares stages 6 and 7. */ +export const TIME_TO_ANSWER_INDEPENDENCE_SAMPLES = 3; +/** + * Stage 6 must not move more than this. The allowance is generous relative to + * the delay: it exists to absorb ordinary run-to-run variance in a number that + * the control cannot legitimately affect, not to permit a shared clock. + */ +export const TIME_TO_ANSWER_INDEPENDENCE_STAGE_SIX_TOLERANCE_MS = 30; +/** Stage 7 must absorb at least this share of the injected delay. */ +export const TIME_TO_ANSWER_INDEPENDENCE_STAGE_SEVEN_SHARE = 0.7; +export interface StageSixSevenIndependence { + fixture: string; + delayMs: number; + normalStageSixMs: readonly number[]; + normalStageSevenMs: readonly number[]; + controlStageSixMs: readonly number[]; + controlStageSevenMs: readonly number[]; + stageSixMedianMs: number; + stageSixControlMedianMs: number; + stageSevenMedianMs: number; + stageSevenControlMedianMs: number; + stageSixDeltaMs: number; + stageSevenDeltaMs: number; +} + +function assertSampleSet(label: string, values: readonly number[]): void { + if (values.length === 0) throw new Error(`${label} requires at least one sample`); + if (values.some((value) => typeof value !== "number" || !Number.isFinite(value) || value < 0)) { + throw new Error(`${label} requires finite non-negative samples`); + } +} + +/** + * Enforce the independence control. Throws when the injected presentation delay + * fails to move stage 7 (the seam is measuring something other than + * presentation) or when it also moves stage 6 (both stages share a clock). A + * capture that cannot demonstrate this must not produce a baseline. + */ +export function assertStageSixSevenIndependence(input: { + fixture: string; + delayMs?: number; + normalStageSixMs: readonly number[]; + normalStageSevenMs: readonly number[]; + controlStageSixMs: readonly number[]; + controlStageSevenMs: readonly number[]; +}): StageSixSevenIndependence { + const delayMs = input.delayMs ?? TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS; + if (!Number.isFinite(delayMs) || delayMs <= 0) { + throw new Error("the independence control requires a positive injected delay"); + } + assertSampleSet("stage 6 normal samples", input.normalStageSixMs); + assertSampleSet("stage 7 normal samples", input.normalStageSevenMs); + assertSampleSet("stage 6 control samples", input.controlStageSixMs); + assertSampleSet("stage 7 control samples", input.controlStageSevenMs); + + const stageSixMedianMs = median(input.normalStageSixMs); + const stageSixControlMedianMs = median(input.controlStageSixMs); + const stageSevenMedianMs = median(input.normalStageSevenMs); + const stageSevenControlMedianMs = median(input.controlStageSevenMs); + const stageSixDeltaMs = stageSixControlMedianMs - stageSixMedianMs; + const stageSevenDeltaMs = stageSevenControlMedianMs - stageSevenMedianMs; + + const stageSixAllowance = Math.max( + TIME_TO_ANSWER_INDEPENDENCE_STAGE_SIX_TOLERANCE_MS, + stageSixMedianMs * 0.25 + ); + if (stageSixDeltaMs > stageSixAllowance) { + throw new Error( + `stage 6 moved by ${stageSixDeltaMs.toFixed(1)} ms under an artificial delay injected AFTER acceptance ` + + `(allowance ${stageSixAllowance.toFixed(1)} ms): stage 6 is not independent of presentation` + ); + } + const requiredStageSevenDelta = delayMs * TIME_TO_ANSWER_INDEPENDENCE_STAGE_SEVEN_SHARE; + if (stageSevenDeltaMs < requiredStageSevenDelta) { + throw new Error( + `stage 7 only moved by ${stageSevenDeltaMs.toFixed(1)} ms for a ${delayMs} ms artificial delay ` + + `(required at least ${requiredStageSevenDelta.toFixed(1)} ms): stage 7 does not measure presentation of the accepted model` + ); + } + if (input.controlStageSevenMs.some((value) => value < delayMs)) { + throw new Error("a control stage-7 sample is shorter than the injected delay, so the delay was not applied"); + } + return { + fixture: input.fixture, + delayMs, + normalStageSixMs: input.normalStageSixMs, + normalStageSevenMs: input.normalStageSevenMs, + controlStageSixMs: input.controlStageSixMs, + controlStageSevenMs: input.controlStageSevenMs, + stageSixMedianMs, + stageSixControlMedianMs, + stageSevenMedianMs, + stageSevenControlMedianMs, + stageSixDeltaMs, + stageSevenDeltaMs + }; +} diff --git a/apps/web/src/lib/performance/timeToAnswerIndependence.test.ts b/apps/web/src/lib/performance/timeToAnswerIndependence.test.ts new file mode 100644 index 00000000..b85d44a2 --- /dev/null +++ b/apps/web/src/lib/performance/timeToAnswerIndependence.test.ts @@ -0,0 +1,105 @@ +import { describe, expect, it } from "vitest"; +import { + TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS, + assertStageSixSevenIndependence +} from "./timeToAnswerEvidence"; + +/** + * Stage 6/7 independence control (#335). + * + * Stage 6 ends when the application accepts a coherent model; stage 7 begins at + * that instant and ends when the accepted model's expected Home content has + * rendered. These tests pin the decision rule the capture enforces before it may + * produce a baseline: an artificial presentation delay injected AFTER acceptance + * must leave stage 6 alone and must move stage 7 by the injected amount. + */ +const NORMAL_STAGE_SIX = [12.4, 13.1, 11.8, 12.9, 12.2, 13.4, 12.0, 12.7, 13.0, 12.5]; +const NORMAL_STAGE_SEVEN = [4.1, 5.2, 3.8, 4.6, 5.0, 4.2, 3.9, 4.8, 4.4, 4.7]; + +const control = (offsetMs: number) => NORMAL_STAGE_SEVEN.map((value) => value + offsetMs); + +describe("stage 6/7 independence control", () => { + it("accepts a control whose injected delay moves stage 7 and leaves stage 6 alone", () => { + const verdict = assertStageSixSevenIndependence({ + fixture: "reference-25", + normalStageSixMs: NORMAL_STAGE_SIX, + normalStageSevenMs: NORMAL_STAGE_SEVEN, + controlStageSixMs: NORMAL_STAGE_SIX.map((value) => value + 2), + controlStageSevenMs: control(TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS) + }); + expect(verdict.stageSevenDeltaMs).toBeCloseTo(TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS, 5); + expect(verdict.stageSixDeltaMs).toBeLessThanOrEqual(30); + }); + + it("rejects a seam whose stage 7 ignores the delayed presentation", () => { + // The RED case: a delayed render that does not move stage 7 means stage 7 is + // not measuring presentation of the accepted model (for example it is the + // same clock as stage 6, or it ends on the notification). + expect(() => + assertStageSixSevenIndependence({ + fixture: "reference-25", + normalStageSixMs: NORMAL_STAGE_SIX, + normalStageSevenMs: NORMAL_STAGE_SEVEN, + controlStageSixMs: NORMAL_STAGE_SIX, + controlStageSevenMs: NORMAL_STAGE_SEVEN + }) + ).toThrow("stage 7 only moved by"); + }); + + it("rejects a seam whose stage 6 moves with the delayed presentation", () => { + expect(() => + assertStageSixSevenIndependence({ + fixture: "reference-25", + normalStageSixMs: NORMAL_STAGE_SIX, + normalStageSevenMs: NORMAL_STAGE_SEVEN, + controlStageSixMs: NORMAL_STAGE_SIX.map((value) => value + 200), + controlStageSevenMs: control(TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS) + }) + ).toThrow("stage 6 is not independent of presentation"); + }); + + it("rejects a control where the delay was not actually applied", () => { + // Stage 7 moved slightly (noise) but a sample is still shorter than the + // injected delay, which can only mean the delay never reached the page. + expect(() => + assertStageSixSevenIndependence({ + fixture: "docker-topology-change", + normalStageSixMs: NORMAL_STAGE_SIX, + normalStageSevenMs: NORMAL_STAGE_SEVEN, + controlStageSixMs: NORMAL_STAGE_SIX, + controlStageSevenMs: [220, 300, 310] + }) + ).toThrow("shorter than the injected delay"); + }); + + it("rejects a partial delay absorption below the reviewed share", () => { + expect(() => + assertStageSixSevenIndependence({ + fixture: "reference-100", + normalStageSixMs: NORMAL_STAGE_SIX, + normalStageSevenMs: NORMAL_STAGE_SEVEN, + controlStageSixMs: NORMAL_STAGE_SIX, + controlStageSevenMs: control(TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS * 0.5) + }) + ).toThrow("stage 7 only moved by"); + }); + + it("refuses to judge an empty, malformed or delay-free control", () => { + const base = { + fixture: "reference-250", + normalStageSixMs: NORMAL_STAGE_SIX, + normalStageSevenMs: NORMAL_STAGE_SEVEN, + controlStageSixMs: NORMAL_STAGE_SIX, + controlStageSevenMs: control(TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS) + }; + expect(() => assertStageSixSevenIndependence({ ...base, controlStageSevenMs: [] })).toThrow( + "requires at least one sample" + ); + expect(() => + assertStageSixSevenIndependence({ ...base, controlStageSixMs: [Number.NaN] }) + ).toThrow("finite non-negative samples"); + expect(() => assertStageSixSevenIndependence({ ...base, delayMs: 0 })).toThrow( + "positive injected delay" + ); + }); +}); diff --git a/apps/web/src/lib/performance/timeToAnswerPollPhase.test.ts b/apps/web/src/lib/performance/timeToAnswerPollPhase.test.ts new file mode 100644 index 00000000..bb73e89a --- /dev/null +++ b/apps/web/src/lib/performance/timeToAnswerPollPhase.test.ts @@ -0,0 +1,303 @@ +/** + * RED-checks for the stage-5 deterministic poll-phase design (#335, methodology + * revision 2). + * + * The design replaced a uniform random trigger delay with a declared phase grid, + * because baseline 3's raw samples proved the jitter did not move the phase: the + * reference-25 samples occupied a 120 ms band of a 2000 ms interval. These tests + * pin the design (grid shape, assignment, bucketing, normalisation) and, above + * all, pin the guards: a sweep confined to a narrow band, a missing phase, an + * uncontrolled publication, an observation that did not travel the real API poller + * path, or a sample that observed no new revision must all be REJECTED. + */ +import { describe, expect, it } from "vitest"; +import { + POLL_PHASE_CONTROL_TOLERANCE_MS, + POLL_PHASE_DIVISIONS, + POLL_PHASE_MIN_SAMPLES_PER_PHASE, + assertFreeRunningPhaseSamples, + assertControlledPhaseEvidence, + assertPollPhaseSweep, + declaredPhaseForSample, + intendedLatencyMs, + observedPhaseBucketMs, + phaseMediansMs, + phaseNormalizedP95Ms, + pollPhaseGridMs, + type PollPhaseSweep +} from "./timeToAnswerPollPhase"; +import { TIME_TO_ANSWER_CONTROLLED_RUNS, TIME_TO_ANSWER_WARMED_SAMPLES } from "./timeToAnswerEvidence"; + +const INTERVAL = 2000; + +/** + * A sweep that satisfies the declared design: every phase present, each run + * sweeping the grid ascending, observed latency equal to the intended value with a + * small, deterministic error so the samples are not identical. + */ +function goodSweep(intervalMs = INTERVAL, errorMs = 4): PollPhaseSweep[] { + const samples: PollPhaseSweep[] = []; + for (let run = 0; run < TIME_TO_ANSWER_CONTROLLED_RUNS; run += 1) { + for (let index = 0; index < POLL_PHASE_DIVISIONS; index += 1) { + const declaredPhaseMs = declaredPhaseForSample(run, index, intervalMs); + const intended = intendedLatencyMs(declaredPhaseMs, intervalMs); + const observed = intended + errorMs + run; + samples.push({ + runIndex: run, + sampleIndex: index, + declaredPhaseMs, + intendedPhaseMs: declaredPhaseMs, + observedPhaseMs: declaredPhaseMs, + intendedLatencyMs: intended, + connectedAtMs: 0, + predictedPublicationAtMs: 1000, + observedPublicationAtMs: 1000, + observedObservationAtMs: 1000 + observed, + observedLatencyMs: observed, + publicationLatencyMs: observed, + observedPhaseBucketMs: observedPhaseBucketMs(observed, intervalMs), + phaseErrorMs: observed - intended, + observedVia: "api-sse", + observedRevision: `rev-${run}-${index}`, + previousRevision: `rev-${run}-${index}-prev`, + phaseControlled: true + }); + } + } + return samples; +} + +describe("stage-5 declared phase grid", () => { + it("INVALIDATES a controlled cell when the requested publication phase was not established", () => { + expect(() => assertControlledPhaseEvidence({ + intendedPhaseMs: 500, + observedPhaseMs: 900, + publicationLatencyMs: 1500, + phaseErrorMs: 0 + })).toThrow(/could not be established; the cell is invalidated/); + }); + + it("records the complete controlled-phase audit evidence", () => { + const sample = goodSweep()[0]!; + expect(sample).toMatchObject({ + intendedPhaseMs: sample.declaredPhaseMs, + observedPhaseMs: sample.declaredPhaseMs, + publicationLatencyMs: sample.observedLatencyMs, + phaseErrorMs: sample.observedLatencyMs - sample.intendedLatencyMs + }); + }); + + it("divides the poll interval into the declared number of phases", () => { + const grid = pollPhaseGridMs(INTERVAL); + expect(grid).toHaveLength(POLL_PHASE_DIVISIONS); + expect([...grid].sort((left, right) => left - right)).toEqual(grid); + expect(new Set(grid).size).toBe(grid.length); + // Every phase sits strictly inside the interval: a publication landing exactly + // on a tick is inherently ambiguous and is deliberately not declared. + expect(grid[0]!).toBeGreaterThan(0); + expect(grid[grid.length - 1]!).toBeLessThan(INTERVAL); + // The declared phases cover essentially the whole interval. + const span = grid[grid.length - 1]! - grid[0]!; + expect(span / INTERVAL).toBeGreaterThan(0.8); + }); + + it("maps each declared phase to a strictly decreasing intended latency", () => { + const latencies = pollPhaseGridMs(INTERVAL).map((phase) => intendedLatencyMs(phase, INTERVAL)); + for (let index = 1; index < latencies.length; index += 1) { + expect(latencies[index]!).toBeLessThan(latencies[index - 1]!); + } + expect(latencies[0]!).toBeLessThan(INTERVAL); + expect(latencies[latencies.length - 1]!).toBeGreaterThan(0); + }); + + it("walks the declared grid once per run and repeats it, identically across runs", () => { + const grid = pollPhaseGridMs(INTERVAL); + for (let index = 0; index < TIME_TO_ANSWER_WARMED_SAMPLES; index += 1) { + const runZero = declaredPhaseForSample(0, index, INTERVAL); + expect(declaredPhaseForSample(1, index, INTERVAL)).toBe(runZero); + expect(declaredPhaseForSample(2, index, INTERVAL)).toBe(runZero); + // Fifteen samples against ten divisions: each run covers the whole grid and then + // repeats its opening phases, so no phase is left with a single sample. + expect(runZero).toBe(grid[index % POLL_PHASE_DIVISIONS]); + } + const coverage = new Map(); + for (let run = 0; run < 3; run += 1) { + for (let index = 0; index < TIME_TO_ANSWER_WARMED_SAMPLES; index += 1) { + const phase = declaredPhaseForSample(run, index, INTERVAL); + coverage.set(phase, (coverage.get(phase) ?? 0) + 1); + } + } + expect(coverage.size).toBe(POLL_PHASE_DIVISIONS); + expect(Math.min(...coverage.values())).toBeGreaterThanOrEqual( + POLL_PHASE_MIN_SAMPLES_PER_PHASE + ); + expect(() => declaredPhaseForSample(0, -1, INTERVAL)).toThrow(); + expect(() => declaredPhaseForSample(0, 1.5, INTERVAL)).toThrow(); + }); + + it("buckets an observed latency to the nearest declared latency", () => { + const grid = pollPhaseGridMs(INTERVAL).map((phase) => intendedLatencyMs(phase, INTERVAL)); + for (const candidate of grid) { + expect(observedPhaseBucketMs(candidate + 3, INTERVAL)).toBe(candidate); + expect(observedPhaseBucketMs(candidate - 3, INTERVAL)).toBe(candidate); + } + }); +}); + +describe("stage-5 phase sweep validity", () => { + it("accepts a sweep that covers the declared grid with a controlled phase", () => { + const verdict = assertPollPhaseSweep(goodSweep(), INTERVAL); + expect(verdict.divisions).toBe(POLL_PHASE_DIVISIONS); + expect(verdict.samples).toBe(POLL_PHASE_DIVISIONS * TIME_TO_ANSWER_CONTROLLED_RUNS); + expect(verdict.samplesPerPhase.every((count) => count >= POLL_PHASE_MIN_SAMPLES_PER_PHASE)).toBe(true); + expect(verdict.spanShare).toBeGreaterThan(0.5); + expect(verdict.directionShare).toBeGreaterThan(0.5); + expect(verdict.worstPhaseErrorMs).toBeLessThanOrEqual(POLL_PHASE_CONTROL_TOLERANCE_MS); + }); + + it("REJECTS a narrow-band sweep — the defect that invalidated baseline 3", () => { + // Every sample clustered near one latency: the sample set a random-jitter + // harness produced for reference-25 (120 ms of a 2000 ms interval). Which guard + // fires first depends on the shape — the declared design is contradicted either + // because the phase was not driven or because the spread is too small — but it + // is always REJECTED. + const narrow = goodSweep().map((sample) => ({ ...sample, observedLatencyMs: 800 + (sample.sampleIndex % 5) })); + expect(() => assertPollPhaseSweep(narrow, INTERVAL)).toThrow( + /narrow band|phase response|not controlled/ + ); + }); + + it("REJECTS a sweep that does not show a phase response", () => { + const flat = goodSweep(INTERVAL, 0).map((sample) => ({ ...sample, observedLatencyMs: 900 })); + expect(() => assertPollPhaseSweep(flat, INTERVAL)).toThrow( + /narrow band|phase response|not controlled/ + ); + }); + + it("REJECTS a missing or thin declared phase", () => { + // One run is not a sweep: every declared phase would have a single sample. + const singleRun = goodSweep().filter((sample) => sample.runIndex === 0); + expect(() => assertPollPhaseSweep(singleRun, INTERVAL)).toThrow(/declared phase|thin/); + // A run that lost a phase fails structurally rather than silently reindexing. + const gap = goodSweep().filter((sample) => !(sample.runIndex === 0 && sample.sampleIndex === 3)); + expect(() => assertPollPhaseSweep(gap, INTERVAL)).toThrow(/declared phase/); + }); + + it("REJECTS a publication phase that was not controlled", () => { + const uncontrolled = goodSweep().map((sample) => + sample.sampleIndex === 7 + ? { ...sample, observedLatencyMs: sample.intendedLatencyMs + 300, phaseErrorMs: 300 } + : sample + ); + expect(() => assertPollPhaseSweep(uncontrolled, INTERVAL)).toThrow(/not controlled/); + }); + + it("REJECTS an observation that did not travel the real API poller path", () => { + const shortcut = goodSweep().map((sample) => + sample.sampleIndex === 2 + ? { ...sample, observedVia: "document-title-poll" } + : sample + ); + expect(() => assertPollPhaseSweep(shortcut, INTERVAL)).toThrow(/real API poller path/); + }); + + it("REJECTS a sample that observed no new revision", () => { + const idle = goodSweep().map((sample) => + sample.sampleIndex === 5 ? { ...sample, observedRevision: sample.previousRevision } : sample + ); + expect(() => assertPollPhaseSweep(idle, INTERVAL)).toThrow(/not a publication observation/); + }); + + it("REJECTS a publication that does not correspond to the controlled trigger", () => { + const unrelated = goodSweep().map((sample) => + sample.sampleIndex === 1 + ? { ...sample, observedPublicationAtMs: sample.predictedPublicationAtMs - INTERVAL } + : sample + ); + expect(() => assertPollPhaseSweep(unrelated, INTERVAL)).toThrow(/controlled trigger/); + }); + + it("REJECTS a declared phase that is not on the declared grid", () => { + const offGrid = goodSweep().map((sample) => + sample.sampleIndex === 4 ? { ...sample, declaredPhaseMs: 12.5 } : sample + ); + expect(() => assertPollPhaseSweep(offGrid, INTERVAL)).toThrow(/declared grid|design requires/); + }); +}); + +describe("stage-5 free-running cells", () => { + const freeRunning = (count: number, latencyMs: (index: number) => number): PollPhaseSweep[] => + Array.from({ length: count }, (_, index) => ({ + ...goodSweep()[0]!, + runIndex: 0, + sampleIndex: index % POLL_PHASE_DIVISIONS, + declaredPhaseMs: 0, + intendedLatencyMs: latencyMs(index), + observedLatencyMs: latencyMs(index), + phaseErrorMs: 0, + observedRevision: `free-${index}`, + previousRevision: `free-${index}-prev`, + phaseControlled: false + })); + + it("accepts real poller observations with the achieved phase recorded", () => { + const verdict = assertFreeRunningPhaseSamples(freeRunning(30, (index) => 100 + index * 50), 30); + expect(verdict.samples).toBe(30); + expect(verdict.spanMs).toBeGreaterThan(0); + expect(verdict.observedPhaseBucketsMs).toHaveLength(30); + }); + + it("REJECTS a free-running cell with too few samples", () => { + expect(() => assertFreeRunningPhaseSamples(freeRunning(5, () => 500), 10)).toThrow(/at least 10 samples/); + }); + + it("REJECTS a free-running sample that saw no new revision or a non-poller path", () => { + const idle = freeRunning(30, () => 500).map((sample, index) => + index === 3 ? { ...sample, observedRevision: sample.previousRevision } : sample + ); + expect(() => assertFreeRunningPhaseSamples(idle, 10)).toThrow(/no new revision/); + const shortcut = freeRunning(30, () => 500).map((sample, index) => + index === 7 ? { ...sample, observedVia: "daemon-direct" } : sample + ); + expect(() => assertFreeRunningPhaseSamples(shortcut, 10)).toThrow(/real API poller path/); + }); + + it("REJECTS a declared sweep over free-running samples", () => { + // The two guards must not be interchangeable: a cell whose phase the harness did + // not drive cannot claim the sweep's coverage. + expect(() => assertPollPhaseSweep(freeRunning(30, () => 500), INTERVAL)).toThrow(/did not drive/); + }); +}); + +describe("stage-5 phase-normalized summary", () => { + it("reports a median per declared phase and normalises over the uniform grid", () => { + const grid = pollPhaseGridMs(INTERVAL); + const runs = [10, 20, 30].map((offset) => + grid.map((phase) => intendedLatencyMs(phase, INTERVAL) + offset) + ); + const medians = phaseMediansMs(runs, INTERVAL); + expect(medians).toHaveLength(POLL_PHASE_DIVISIONS); + // Each phase's median is its 20 ms sample, and the phases are ordered by + // intended latency: the earliest declared phase is the SLOWEST. + expect(medians[0]).toBeGreaterThan(medians[medians.length - 1]!); + const normalized = phaseNormalizedP95Ms(runs, INTERVAL); + // Nearest-rank p95 over 15 phase medians is the largest phase median, so the + // normalized figure equals the slowest declared phase's median. + expect(normalized).toBe(Math.max(...medians)); + expect(normalized).toBe(medians[0]); + }); + + it("uses every repeated positional sample for its declared phase", () => { + // Samples 10–14 repeat phases 0–4. Their deliberately large values make a + // first-occurrence-only implementation observably wrong. + const runs = Array.from({ length: TIME_TO_ANSWER_CONTROLLED_RUNS }, (_, run) => + Array.from({ length: TIME_TO_ANSWER_WARMED_SAMPLES }, (_, sample) => + sample < POLL_PHASE_DIVISIONS ? sample * 10 + run : 1_000 + (sample - POLL_PHASE_DIVISIONS) * 10 + run + ) + ); + const medians = phaseMediansMs(runs, INTERVAL); + expect(medians.slice(0, 5)).toEqual([501, 511, 521, 531, 541]); + expect(medians.slice(5)).toEqual([51, 61, 71, 81, 91]); + expect(phaseNormalizedP95Ms(runs, INTERVAL)).toBe(541); + }); +}); diff --git a/apps/web/src/lib/performance/timeToAnswerPollPhase.ts b/apps/web/src/lib/performance/timeToAnswerPollPhase.ts new file mode 100644 index 00000000..5ac2c4a1 --- /dev/null +++ b/apps/web/src/lib/performance/timeToAnswerPollPhase.ts @@ -0,0 +1,478 @@ +/** + * Stage-5 poll-phase design (#335, methodology revision 2). + * + * Stage 5 measures: authoritative daemon publication -> the Node/SSE layer + * observes that revision, through the CURRENT production polling mechanism + * (a fixed `DOCKERMAP_SSE_INTERVAL_MS` poller; 2000 ms today). + * + * The mechanism's latency is a sawtooth in the phase of the publication within a + * poll interval, so a capture that lets the phase fall where it likes cannot + * characterise it: baseline 3's reference-25 samples occupied a 120 ms band + * (6.0 % of the interval) even though the harness slept a uniform sub-interval + * delay before each trigger, because both the daemon's refresh loop and the + * API's poller are fixed 2 s loops and the measured gap was their phase offset. + * + * This module therefore declares an EXPLICIT deterministic phase sweep instead + * of an assumed uniform random distribution: + * + * - the poll interval is divided into `POLL_PHASE_DIVISIONS` equal parts, giving + * one declared phase per recorded sample per run (15 phases, 15 samples); + * - phase `p` places the publication at `(p + 0.5) * interval / divisions`, so the + * intended latency is `interval - that offset`: the sweep covers the interval + * from half a division to `divisions - 0.5` divisions, and no phase sits on a + * poll tick boundary, where a publication would be inherently ambiguous; + * - each controlled run sweeps the phases in ascending order, so each phase has + * exactly `TIME_TO_ANSWER_CONTROLLED_RUNS` samples per cell and the phase of + * every raw sample is recoverable from its position in the run; + * - the phase is DRIVEN, not hoped for: the harness connects its observation + * stream at a computed instant so that the next poll tick after the predicted + * publication falls at the intended latency, then verifies the result. + * + * The summary derived from this design is a PHASE-NORMALIZED figure: it weights + * the declared phases uniformly to characterise the latency imposed by the fixed + * polling mechanism. It is not an observed user-traffic distribution and not + * network latency. + */ + +/** Equal parts the poll interval is divided into; one declared phase per sample per run. */ +/** + * Declared divisions of the poll interval. Ten gives a 200 ms grid step, which is + * what the mechanism allows: the observation tick is a Node timer that drifts under + * load (measured +52 ms at 250 containers), so the step must exceed the achievable + * control precision by a margin or neighbouring phases blur together. Ten divisions + * still sweep 90% of the interval (latencies 100–1900 ms). + */ +export const POLL_PHASE_DIVISIONS = 10; + +/** + * How far an observed sample may sit from its declared phase before the cell is + * invalid. The grid spacing is `interval / divisions` (133.3 ms at 2000 ms), so + * the tolerance keeps adjacent phases distinguishable while absorbing the few + * milliseconds of publication-grid drift and detection delay. + */ +/** + * Declared control tolerance: how far the observed latency may sit from the intended + * one before a controlled sample is rejected as uncontrolled. Half the grid step + * (100 ms) is the mathematical limit — anything larger could bucket into a + * neighbouring phase — and 90 ms leaves room for the poll timer's real drift. + */ +export const POLL_PHASE_CONTROL_TOLERANCE_MS = 90; + +/** + * The observed sweep must span at least this share of the poll interval, and the + * smallest publication offset must be at least this much slower than the largest + * one. A sweep confined to a narrow band cannot satisfy either, which is exactly + * the defect that invalidated baseline 3's reference cells. + */ +export const POLL_PHASE_MIN_SPAN_SHARE = 0.5; +export const POLL_PHASE_MIN_DIRECTION_SHARE = 0.5; + +/** Every declared phase must appear at least this many times in a controlled cell. */ +export const POLL_PHASE_MIN_SAMPLES_PER_PHASE = 2; + +/** + * Cells whose publication the harness can actually trigger, and which therefore get + * the declared phase sweep. + * + * The two provider-state fixtures are NOT here on purpose. Their revisions advance + * from the daemon's own host provider collection — the harness has no input that + * makes the daemon publish at a chosen instant — so their stage-5 samples are + * FREE-RUNNING: the phase each sample achieved is recorded, the sweep's coverage and + * direction guards do not apply, and no phase-normalized figure is derived for them. + * They still measure today's real poll wait, and they are still declared in the + * matrix; what they cannot do is place the publication. + */ +export const POLL_PHASE_CONTROLLED_FIXTURES = [ + "reference-25", + "reference-100", + "reference-250", + "docker-topology-change" +] as const; + +export function isPhaseControlledFixture(fixture: string): boolean { + return (POLL_PHASE_CONTROLLED_FIXTURES as readonly string[]).includes(fixture); +} + +export interface PollPhaseSweep { + runIndex: number; + sampleIndex: number; + /** Declared publication offset after the enclosing poll tick. */ + declaredPhaseMs: number; + /** Explicit audit name for the phase the harness requested. */ + intendedPhaseMs: number; + /** Publication phase actually witnessed from the connected observation stream. */ + observedPhaseMs: number; + /** Latency the declared phase should produce. */ + intendedLatencyMs: number; + /** When the harness connected its observation stream, relative to the run clock. */ + connectedAtMs: number; + /** Predicted publication instant for this sample, relative to the run clock. */ + predictedPublicationAtMs: number; + /** Observed publication instant, relative to the run clock. */ + observedPublicationAtMs: number; + /** Observed poll tick that carried the revision, relative to the run clock. */ + observedObservationAtMs: number; + observedLatencyMs: number; + /** Publication-to-observation latency on the real API poller path. */ + publicationLatencyMs: number; + /** The declared phase whose intended latency is nearest to the observed latency. */ + observedPhaseBucketMs: number; + /** Observed minus intended latency. */ + phaseErrorMs: number; + /** How the observation reached the harness. Only the real poller path is valid. */ + observedVia: string; + /** The revision observed, and the revision that was current before the sample. */ + observedRevision: string; + previousRevision: string; + /** + * Whether the harness drove this sample's publication phase. Free-running cells + * (the provider-state fixtures) record the phase they achieved instead, and are + * excluded from the sweep's coverage and direction guards. + */ + phaseControlled: boolean; +} + +/** + * Fail closed before a controlled sample is recorded. A phase-controlled cell + * may only claim control when the witnessed publication phase and poll latency + * both match the requested phase within the declared tolerance. + */ +export function assertControlledPhaseEvidence(evidence: Pick): void { + const values = { + intendedPhaseMs: evidence.intendedPhaseMs, + observedPhaseMs: evidence.observedPhaseMs, + publicationLatencyMs: evidence.publicationLatencyMs, + phaseErrorMs: evidence.phaseErrorMs + }; + for (const [name, value] of Object.entries(values)) { + if (!Number.isFinite(value)) throw new Error(`stage-5 ${name} is not finite`); + } + if (Math.abs(evidence.observedPhaseMs - evidence.intendedPhaseMs) > POLL_PHASE_CONTROL_TOLERANCE_MS) { + throw new Error("stage-5 requested publication phase could not be established; the cell is invalidated"); + } + if (Math.abs(evidence.phaseErrorMs) > POLL_PHASE_CONTROL_TOLERANCE_MS) { + throw new Error("stage-5 requested poll phase could not be established; the cell is invalidated"); + } +} + +export interface PollPhaseValidity { + divisions: number; + intervalMs: number; + samples: number; + declaredPhasesMs: readonly number[]; + observedPhasesMs: readonly number[]; + minObservedLatencyMs: number; + maxObservedLatencyMs: number; + spanMs: number; + spanShare: number; + slowestPhaseMedianMs: number; + fastestPhaseMedianMs: number; + directionShare: number; + worstPhaseErrorMs: number; + samplesPerPhase: readonly number[]; +} + +function assertInterval(intervalMs: number): void { + if (!Number.isFinite(intervalMs) || intervalMs <= 0) { + throw new Error("the poll interval must be a finite positive number of milliseconds"); + } +} + +/** + * Declared publication offsets within the poll interval, ascending. + * + * Phase `p` places the publication at `(p + 0.5) * interval / divisions`, so every + * declared phase sits at the CENTRE of its division: never on a poll tick boundary + * (where a publication is inherently ambiguous) and never at the very end of the + * interval. + */ +export function pollPhaseGridMs(intervalMs: number): number[] { + assertInterval(intervalMs); + const step = intervalMs / POLL_PHASE_DIVISIONS; + return Array.from({ length: POLL_PHASE_DIVISIONS }, (_, index) => (index + 0.5) * step); +} + +/** Declared latency for a publication offset: the wait until the next poll tick. */ +export function intendedLatencyMs(phaseMs: number, intervalMs: number): number { + assertInterval(intervalMs); + if (!Number.isFinite(phaseMs) || phaseMs <= 0 || phaseMs >= intervalMs) { + throw new Error("a declared phase must sit strictly inside the poll interval"); + } + return intervalMs - phaseMs; +} + +/** The declared phase for a sample: run `r` sweeps the grid ascending from index 0. */ +export function declaredPhaseIndexForSample(_runIndex: number, sampleIndex: number): number { + if (!Number.isInteger(sampleIndex) || sampleIndex < 0) { + throw new Error("a stage-5 sample index must be a non-negative integer"); + } + // Fifteen recorded samples against ten declared divisions: each run walks the whole + // grid and then repeats its first five phases, so across the three controlled runs + // every declared phase carries at least three samples. + return sampleIndex % POLL_PHASE_DIVISIONS; +} + +/** The declared phase a raw sample must have produced, from its position in its run. */ +export function declaredPhaseForSample(runIndex: number, sampleIndex: number, intervalMs: number): number { + return pollPhaseGridMs(intervalMs)[declaredPhaseIndexForSample(runIndex, sampleIndex)]!; +} + +/** Nearest declared latency for an observed latency: the observed phase bucket. */ +export function observedPhaseBucketMs(observedLatencyMs: number, intervalMs: number): number { + const grid = pollPhaseGridMs(intervalMs).map((phase) => intendedLatencyMs(phase, intervalMs)); + let best = grid[0]!; + let bestDistance = Number.POSITIVE_INFINITY; + for (const candidate of grid) { + const distance = Math.abs(candidate - observedLatencyMs); + if (distance < bestDistance) { + best = candidate; + bestDistance = distance; + } + } + return best; +} + +function median(values: readonly number[]): number { + const ordered = [...values].sort((left, right) => left - right); + const middle = Math.floor(ordered.length / 2); + return ordered.length % 2 === 1 ? ordered[middle]! : (ordered[middle - 1]! + ordered[middle]!) / 2; +} + +/** + * Nearest-rank p95 over the declared-phase medians: the phase-normalized figure. + * Each declared phase is weighted equally, which is a statement about the + * MECHANISM under uniformly sampled phase offsets — never about observed + * user-traffic distribution, and never about network distance. + */ +export function phaseNormalizedP95Ms(runs: readonly (readonly number[])[], intervalMs: number): number { + const perPhaseMedians = phaseMediansMs(runs, intervalMs); + const ordered = [...perPhaseMedians].sort((left, right) => left - right); + return ordered[Math.ceil(ordered.length * 0.95) - 1]!; +} + +/** Median observed latency per declared phase, ascending by phase. */ +export function phaseMediansMs(runs: readonly (readonly number[])[], intervalMs: number): number[] { + const grid = pollPhaseGridMs(intervalMs); + return grid.map((_phase, phaseIndex) => { + const samples = runs.flatMap((run, runIndex) => + run.filter( + (value, sampleIndex): value is number => + declaredPhaseIndexForSample(runIndex, sampleIndex) === phaseIndex && + typeof value === "number" && + Number.isFinite(value) + ) + ); + if (samples.length === 0) { + throw new Error(`no samples exist for declared phase ${phaseIndex + 1}/${POLL_PHASE_DIVISIONS}`); + } + return median(samples); + }); +} + +/** + * Enforce the declared experimental design on one stage-5 cell. Throws when the + * sweep does not cover the interval, when a declared phase is missing, when the + * phase was not actually driven to its declared offset, when an observation did + * not arrive through the real API poller path, or when the observed spread is + * too narrow to characterise a sawtooth whose range is one poll interval. + */ +export function assertPollPhaseSweep( + samples: readonly PollPhaseSweep[], + intervalMs: number, + minSamplesPerPhase = POLL_PHASE_MIN_SAMPLES_PER_PHASE +): PollPhaseValidity { + assertInterval(intervalMs); + if (samples.length === 0) throw new Error("the stage-5 phase sweep has no samples"); + const grid = pollPhaseGridMs(intervalMs); + + const declaredPhasesMs: number[] = []; + const observedPhasesMs: number[] = []; + const samplesPerPhase = grid.map(() => 0); + let worstPhaseErrorMs = 0; + let minObservedLatencyMs = Number.POSITIVE_INFINITY; + let maxObservedLatencyMs = Number.NEGATIVE_INFINITY; + + for (const sample of samples) { + if (!Number.isFinite(sample.observedLatencyMs) || sample.observedLatencyMs < 0) { + throw new Error("a stage-5 sample has a non-finite observed latency"); + } + if (!sample.phaseControlled) { + throw new Error( + "the declared phase sweep was applied to a sample the harness did not drive; " + + "free-running cells use the free-running guard instead" + ); + } + if (sample.observedVia !== "api-sse") { + throw new Error( + `a stage-5 sample was observed via ${sample.observedVia} instead of the real API poller path` + ); + } + if (!sample.observedRevision || sample.observedRevision === sample.previousRevision) { + throw new Error("a stage-5 sample did not observe a new revision, so it is not a publication observation"); + } + if (sample.observedPublicationAtMs < sample.predictedPublicationAtMs - intervalMs / 2) { + throw new Error("a stage-5 sample's observed publication does not correspond to the controlled trigger"); + } + const expectedPhase = declaredPhaseForSample(sample.runIndex, sample.sampleIndex, intervalMs); + if (Math.abs(expectedPhase - sample.declaredPhaseMs) > 1e-6) { + throw new Error( + `stage-5 sample ${sample.sampleIndex} of run ${sample.runIndex} declares phase ${sample.declaredPhaseMs} ` + + `but the design requires ${expectedPhase}` + ); + } + const phaseErrorMs = sample.observedLatencyMs - sample.intendedLatencyMs; + worstPhaseErrorMs = Math.max(worstPhaseErrorMs, Math.abs(phaseErrorMs)); + if (Math.abs(phaseErrorMs) > POLL_PHASE_CONTROL_TOLERANCE_MS) { + throw new Error( + `stage-5 phase ${sample.declaredPhaseMs.toFixed(1)} ms produced ${sample.observedLatencyMs.toFixed(1)} ms ` + + `instead of ${sample.intendedLatencyMs.toFixed(1)} ms (error ${phaseErrorMs.toFixed(1)} ms exceeds the ` + + `${POLL_PHASE_CONTROL_TOLERANCE_MS} ms tolerance): the publication phase was not controlled` + ); + } + const bucket = observedPhaseBucketMs(sample.observedLatencyMs, intervalMs); + const declaredPhaseIndex = declaredPhaseIndexOfValue(sample.declaredPhaseMs, grid); + if (declaredPhaseIndex < 0) { + throw new Error(`stage-5 declared phase ${sample.declaredPhaseMs} is not on the declared grid`); + } + declaredPhasesMs.push(sample.declaredPhaseMs); + observedPhasesMs.push(bucket); + samplesPerPhase[declaredPhaseIndex] = (samplesPerPhase[declaredPhaseIndex] ?? 0) + 1; + minObservedLatencyMs = Math.min(minObservedLatencyMs, sample.observedLatencyMs); + maxObservedLatencyMs = Math.max(maxObservedLatencyMs, sample.observedLatencyMs); + } + + const sparse = samplesPerPhase + .map((count, index) => ({ count, index })) + .filter((entry) => entry.count < minSamplesPerPhase); + if (sparse.length > 0) { + throw new Error( + `the stage-5 sweep must cover every declared phase at least ${minSamplesPerPhase} times; ` + + `missing or thin phases: ${sparse.map((entry) => entry.index + 1).join(", ")}` + ); + } + + const spanMs = maxObservedLatencyMs - minObservedLatencyMs; + const spanShare = spanMs / intervalMs; + if (spanShare < POLL_PHASE_MIN_SPAN_SHARE) { + throw new Error( + `the stage-5 sweep spans only ${spanMs.toFixed(1)} ms (${(spanShare * 100).toFixed(1)} % of the ` + + `${intervalMs} ms interval): a narrow band cannot characterise the polling mechanism's phase response` + ); + } + + const perPhase = phaseMediansMs( + groupRunsByPhase(samples, intervalMs), + intervalMs + ); + const slowestPhaseMedianMs = perPhase[0]!; + const fastestPhaseMedianMs = perPhase[perPhase.length - 1]!; + const directionShare = (slowestPhaseMedianMs - fastestPhaseMedianMs) / intervalMs; + if (directionShare < POLL_PHASE_MIN_DIRECTION_SHARE) { + throw new Error( + `the stage-5 sweep does not show a phase response: the earliest declared phase median ` + + `(${slowestPhaseMedianMs.toFixed(1)} ms) is only ${(directionShare * 100).toFixed(1)} % of an interval ` + + `above the latest (${fastestPhaseMedianMs.toFixed(1)} ms)` + ); + } + + return { + divisions: POLL_PHASE_DIVISIONS, + intervalMs, + samples: samples.length, + declaredPhasesMs, + observedPhasesMs, + minObservedLatencyMs, + maxObservedLatencyMs, + spanMs, + spanShare, + slowestPhaseMedianMs, + fastestPhaseMedianMs, + directionShare, + worstPhaseErrorMs, + samplesPerPhase + }; +} + +function declaredPhaseIndexOfValue(phaseMs: number, grid: readonly number[]): number { + return grid.findIndex((candidate) => Math.abs(candidate - phaseMs) < 1e-6); +} + +export interface FreeRunningPhaseValidity { + samples: number; + minObservedLatencyMs: number; + maxObservedLatencyMs: number; + spanMs: number; + observedPhaseBucketsMs: readonly number[]; +} + +/** + * Guard for FREE-RUNNING stage-5 cells — the provider-state fixtures, whose + * publications the harness cannot place. Coverage and direction are NOT required + * (the design does not claim to control the phase there), but the sample must still + * be a real observation: a new revision, seen through the real API poller path, with + * the phase it achieved recorded. A cell that recorded no usable samples, or that + * saw a non-poller observation, is rejected. + */ +export function assertFreeRunningPhaseSamples( + samples: readonly PollPhaseSweep[], + minimumSamples = 10 +): FreeRunningPhaseValidity { + if (samples.length < minimumSamples) { + throw new Error( + `a free-running stage-5 cell needs at least ${minimumSamples} samples, got ${samples.length}` + ); + } + let minObservedLatencyMs = Number.POSITIVE_INFINITY; + let maxObservedLatencyMs = Number.NEGATIVE_INFINITY; + const observedPhaseBucketsMs: number[] = []; + for (const sample of samples) { + if (sample.phaseControlled) { + throw new Error("a free-running cell must not contain phase-controlled samples"); + } + if (sample.observedVia !== "api-sse") { + throw new Error("a free-running stage-5 sample did not arrive through the real API poller path"); + } + if (!sample.observedRevision || sample.observedRevision === sample.previousRevision) { + throw new Error("a free-running stage-5 sample observed no new revision"); + } + if (!Number.isFinite(sample.observedLatencyMs) || sample.observedLatencyMs < 0) { + throw new Error("a free-running stage-5 sample has a non-finite observed latency"); + } + observedPhaseBucketsMs.push(sample.observedPhaseBucketMs); + minObservedLatencyMs = Math.min(minObservedLatencyMs, sample.observedLatencyMs); + maxObservedLatencyMs = Math.max(maxObservedLatencyMs, sample.observedLatencyMs); + } + return { + samples: samples.length, + minObservedLatencyMs, + maxObservedLatencyMs, + spanMs: maxObservedLatencyMs - minObservedLatencyMs, + observedPhaseBucketsMs + }; +} + +/** Rebuild per-run sample arrays from sweep records, so the shared math can be reused. */ +function groupRunsByPhase(samples: readonly PollPhaseSweep[], intervalMs: number): number[][] { + const runIndexes = [...new Set(samples.map((sample) => sample.runIndex))].sort((left, right) => left - right); + return runIndexes.map((runIndex) => { + const run = samples + .filter((sample) => sample.runIndex === runIndex) + .sort((left, right) => left.sampleIndex - right.sampleIndex); + const values: number[] = []; + for (const sample of run) { + if (sample.sampleIndex !== values.length) { + throw new Error(`run ${runIndex} has a missing or duplicate declared phase sample index`); + } + assertControlledPhaseEvidence(sample); + // The sample's own recorded phase decides its slot, so the shared math cannot + // silently disagree with the declaration the capture used. + if (Math.abs(sample.declaredPhaseMs - declaredPhaseForSample(runIndex, sample.sampleIndex, intervalMs)) >= 1e-6) { + throw new Error(`run ${runIndex} has an incorrectly declared stage-5 phase`); + } + values.push(sample.observedLatencyMs); + } + return values; + }); +} diff --git a/apps/web/src/lib/performance/timeToAnswerPromotion.test.ts b/apps/web/src/lib/performance/timeToAnswerPromotion.test.ts new file mode 100644 index 00000000..1474467c --- /dev/null +++ b/apps/web/src/lib/performance/timeToAnswerPromotion.test.ts @@ -0,0 +1,474 @@ +/** + * RED-checks for the time-to-answer promotion gate (#335). + * + * Every case here must fail for an EVIDENCE reason — a bad comparison — and not + * because the fixture happens to be malformed in some unrelated way. The + * candidate in each rejection case is otherwise a complete, valid artifact. + */ +import { describe, expect, it } from "vitest"; +import { + TIME_TO_ANSWER_BASELINE, + TIME_TO_ANSWER_CONTROLLED_RUNS, + TIME_TO_ANSWER_MATRIX, + TIME_TO_ANSWER_METHODOLOGY, + TIME_TO_ANSWER_WARMED_SAMPLES, + TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS, + TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS, + TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES, + TIME_TO_ANSWER_WARM_UP_METRICS, + deriveFrozenWarmUpCount, + deriveWarmUpCalibrationReport, + assertTimeToAnswerPromotion, + assertDaemonBinaryProvenance, + compatibleTimeToAnswerEnvironment, + isScenarioCell, + splitWarmedObservations, + summarizeTimeToAnswerStage, + TIME_TO_ANSWER_STAGE_KIND, + TIME_TO_ANSWER_STAGES, + timeToAnswerLimit, + validateTimeToAnswerEvidence +} from "./timeToAnswerEvidence"; + +const environment = { + runnerClass: "linux-x86_64-dedicated", + cpuClass: "cpus-16vcpu", + osImage: "ubuntu-26.04", + osKernel: "7.0.0-31-generic", + nodeRevision: "22.23.2", + rustRevision: "1.88.0", + dockerRevision: "29.8.1", + ssePollIntervalMs: "2000", + daemonBinarySha256: "eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee", + daemonBinaryBuild: "cargo-build-release-locked-p-dockermap-daemon", + cargoRevision: "cargo-1.88.0", + harnessRevision: "dddddddddddddddddddddddddddddddddddddddd", + browserEngine: "chromium", + browserRevision: "1.61.0", + browserFlags: ["--disable-background-networking"], + fontEnvironment: "system-default", + buildMode: "production", + fixtureRevision: "dockermap-v1/time-to-answer-fixtures-1", + sourceRevision: "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + methodologyVersion: TIME_TO_ANSWER_METHODOLOGY +}; + +/** 15 finite non-negative warmed samples with a per-run offset. */ +function samples(base: number): number[] { + return Array.from({ length: TIME_TO_ANSWER_WARMED_SAMPLES }, (_, index) => base + index * 0.1); +} + +function artifact(overrides: { environment?: Record; records?: unknown[] } = {}) { + const provenance = (record: any) => ({ + ...record, + measurementProtocol: record.measurementProtocol ?? protocol(record.fixture, record.stage), + sourceEvidenceFile: record.sourceEvidenceFile ?? evidenceFile(record.fixture, record.stage), + checkpointSha: record.checkpointSha ?? "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + }); + return { + baseline: TIME_TO_ANSWER_BASELINE, + environment: { ...environment, ...(overrides.environment ?? {}) }, + records: + (overrides.records ? overrides.records.map(provenance) : + TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => ({ +fixture, +stage, + measurementProtocol: protocol(fixture, stage), + sourceEvidenceFile: evidenceFile(fixture, stage), +checkpointSha: "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", +runs: [samples(10), samples(11), samples(12)] + }))) + }; +} + +function protocol(fixture: string, stage: string): string { + void fixture; + if (stage === "publicationToNodeObservationMs") return "controlled-poll-phase"; + return "end-to-end"; +} + +function evidenceFile(fixture: string, stage: string): string { + return protocol(fixture, stage) === "controlled-poll-phase" ? "stage-five.raw.json" : "general.raw.json"; +} + +function candidate(overrides: { environment?: Record; records?: unknown[] } = {}) { + return artifact({ + ...overrides, + environment: { sourceRevision: "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", ...(overrides.environment ?? {}) } + }); +} + +describe("time-to-answer promotion gate", () => { + it("uses the declared fixed 60-observation burn-in without a per-metric table", () => { + expect(TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS).toBe(60); + }); + + it("derives the earliest sustained calibration point, maximum fixture count, and fixed margin", () => { + const stable = Array.from({ length: TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS }, () => 10); + // This fixture is unsettled at candidates 2 and 3, then stationary through + // every eligible position. The metric result is its earliest stable point +2. + stable[0] = 100; + stable[1] = 100; + stable[2] = 100; + stable[3] = 100; + const derived = deriveFrozenWarmUpCount([ + { fixture: "reference-25", metric: "dockerObservationMs", observations: Array.from({ length: TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS }, () => 10) }, + { fixture: "reference-100", metric: "dockerObservationMs", observations: stable }, + { fixture: "reference-250", metric: "dockerObservationMs", observations: Array.from({ length: TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS }, () => 10) } + ]); + expect(derived.fixtureCounts).toEqual({ "reference-25": 2, "reference-100": 6, "reference-250": 2 }); + expect(derived.frozenWarmUpCount).toBe(8); + }); + + it("fails rather than extrapolating when the safety margin is not evidence-backed", () => { + const observations = Array.from({ length: TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS }, () => 10); + // Only candidate 45 is stationary, so 45 + the fixed margin exceeds the + // final eligible position (45) and must not become a frozen count. + for (let index = 0; index < 43; index += 1) observations[index] = 100; + expect(() => deriveFrozenWarmUpCount(TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES.map((fixture) => ({ fixture, metric: "dockerObservationMs", observations })))).toThrow("calibration conflict"); + }); + + it("reports every metric on conflict and never makes a partial table authoritative", () => { + const stable = Array.from({ length: TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS }, () => 10); + const conflicted = [...stable]; + // Stable only at the final eligible candidate: +2 then lacks a following 15. + for (let index = 0; index < 43; index += 1) conflicted[index] = 100; + const cells = TIME_TO_ANSWER_WARM_UP_METRICS.flatMap((metric) => TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES.map((fixture) => ({ fixture, metric, observations: metric === "dockerObservationMs" && fixture === "reference-25" ? conflicted : stable }))); + const report = deriveWarmUpCalibrationReport(cells); + expect(report.verdict).toBe("CONFLICT"); + expect(report.metrics).toHaveLength(TIME_TO_ANSWER_WARM_UP_METRICS.length); + expect(report.metrics.find((metric) => metric.metric === "dockerObservationMs")).toMatchObject({ verdict: "CONFLICT", proposedWarmUpCount: 47, requiredEvidenceLength: 62, evidenceBacked: false }); + expect(report.nonAuthoritativeProposedWarmUpCounts).toEqual({}); + expect(report.metrics.find((metric) => metric.metric === "buildModelMs")?.verdict).toBe("PASS"); + }); + + it("records candidate ratios and atomically proposes every count only on complete success", () => { + const observations = Array.from({ length: TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS }, () => 10); + const report = deriveWarmUpCalibrationReport(TIME_TO_ANSWER_WARM_UP_METRICS.flatMap((metric) => TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES.map((fixture) => ({ fixture, metric, observations })))); + expect(report.verdict).toBe("PASS"); + expect(Object.keys(report.nonAuthoritativeProposedWarmUpCounts)).toEqual(TIME_TO_ANSWER_WARM_UP_METRICS); + expect(report.metrics[0]?.fixtures[0]?.stableWarmUpCount).toBe(2); + expect(report.metrics[0]?.fixtures[0]?.candidates[0]).toMatchObject({ candidate: 2, ratio: 1, inBand: true, sustained: true }); + }); + + it("rejects a calibration that omits or adds a reference fixture", () => { + const observations = Array.from({ length: TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS }, () => 10); + expect(() => deriveFrozenWarmUpCount([ + { fixture: "reference-25", metric: "dockerObservationMs", observations }, + { fixture: "reference-100", metric: "dockerObservationMs", observations } + ])).toThrow("every declared reference fixture"); + expect(() => deriveFrozenWarmUpCount([ + ...TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES.map((fixture) => ({ fixture, metric: "dockerObservationMs", observations })), + { fixture: "docker-topology-change", metric: "dockerObservationMs", observations } + ])).toThrow("every declared reference fixture"); + }); + + it("accepts a compatible candidate inside the reviewed budget", () => { + expect(() => assertTimeToAnswerPromotion(artifact(), candidate())).not.toThrow(); + }); + + it("accepts the provenance differences a candidate may carry: source revision, harness revision, rebuilt digest", () => { + // Source and harness revisions differ by construction, and the daemon binary is + // REBUILT from the candidate checkout — so a byte-identical digest is not even + // reproducible across a changed CARGO_HOME. Requiring any of them to match would + // make every candidate that touches the product or the harness uncomparable. + const comparable = candidate({ + environment: { + sourceRevision: "cccccccccccccccccccccccccccccccccccccccc", + harnessRevision: "ab".repeat(20), + daemonBinarySha256: "f".repeat(64) + } + }); + expect(() => assertTimeToAnswerPromotion(artifact(), comparable)).not.toThrow(); + const baseline = validateTimeToAnswerEvidence(artifact()); + const validated = validateTimeToAnswerEvidence(comparable); + expect(compatibleTimeToAnswerEnvironment(baseline.environment, validated.environment)).toBe(true); + expect(validated.environment.daemonBinarySha256).toBe("f".repeat(64)); + }); + + it("rejects a candidate measured under a different methodology version", () => { + const other = candidate({ environment: { methodologyVersion: "dockermap-v1/time-to-answer-methodology-1" } }); + expect( + compatibleTimeToAnswerEnvironment( + validateTimeToAnswerEvidence(artifact()).environment, + validateTimeToAnswerEvidence(other).environment + ) + ).toBe(false); + expect(() => assertTimeToAnswerPromotion(artifact(), other)).toThrow( + "does not match the pinned baseline environment" + ); + }); + + it("rejects a candidate above the reviewed budget, naming the cell", () => { + const slow = candidate({ + records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => + fixture === "reference-250" && stage === "dockerObservationMs" + ? { fixture, stage, runs: [samples(1_000), samples(1_000), samples(1_000)] } + : { fixture, stage, runs: [samples(10), samples(11), samples(12)] } + ) + }); + expect(() => assertTimeToAnswerPromotion(artifact(), slow)).toThrow("promotion limit"); + // The limit itself is the reviewed rule, not an invented constant. + expect(timeToAnswerLimit(10)).toBe(12.5); + expect(timeToAnswerLimit(1)).toBe(3); + }); + + it("rejects a slow cell the budget tolerates only just", () => { + const limit = timeToAnswerLimit(12); + const pass = candidate({ + records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => + fixture === "reference-100" && stage === "buildModelMs" + ? { fixture, stage, runs: [[limit, limit, limit, ...Array(12).fill(limit)], [limit, limit, limit, ...Array(12).fill(limit)], [limit, limit, limit, ...Array(12).fill(limit)]] } + : { fixture, stage, runs: [samples(1), samples(1), samples(1)] } + ) + }); + expect(() => assertTimeToAnswerPromotion(artifact(), pass)).not.toThrow(); + const fail = candidate({ + records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => + fixture === "reference-100" && stage === "buildModelMs" + ? { fixture, stage, runs: [[limit + 1, ...Array(14).fill(limit + 1)], [limit + 1, ...Array(14).fill(limit + 1)], [limit + 1, ...Array(14).fill(limit + 1)]] } + : { fixture, stage, runs: [samples(1), samples(1), samples(1)] } + ) + }); + expect(() => assertTimeToAnswerPromotion(artifact(), fail)).toThrow("promotion limit"); + }); + + it.each([ + ["runner class", "runnerClass", "some-other-runner"], + ["cpu class", "cpuClass", "cpus-2vcpu"], + ["os image", "osImage", "debian-13"], + ["os kernel", "osKernel", "6.8.0-31-generic"], + ["node revision", "nodeRevision", "20.11.0"], + ["rust revision", "rustRevision", "1.80.0"], + ["chromium revision", "browserRevision", "1.50.0"], + ["fixture revision", "fixtureRevision", "dockermap-v1/other-fixtures"], + ["sse poll interval", "ssePollIntervalMs", "1000"], + ["font environment", "fontEnvironment", "different-fonts"] + ])("rejects a candidate whose %s does not match the pinned baseline", (_label, key, value) => { + expect(() => assertTimeToAnswerPromotion(artifact(), candidate({ environment: { [key]: value } }))).toThrow( + "does not match the pinned baseline environment" + ); + }); + + it("treats dockerRevision as informational: a host engine change must not fail a comparison", () => { + // No measured stage exercises the host Docker daemon — the capture runs + // against the deterministic fixture daemon — so pinning it as a + // compatibility key would reject a candidate for an untouched dimension. + expect(() => + assertTimeToAnswerPromotion(artifact(), candidate({ environment: { dockerRevision: "30.1.0" } })) + ).not.toThrow(); + }); + + it("pins the median-of-three aggregation, not the first run or the pooled mean", () => { + // One slow run and two fast runs: the median must pass, while a first-run + // p95 or a pooled mean would exceed the budget. + const slowFirst = candidate({ + records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => + fixture === "reference-250" && stage === "commandQueryMs" + ? { + fixture, + stage, + runs: [ + [...Array(15).fill(500)], + [...Array(15).fill(10)], + [...Array(15).fill(10)] + ] + } + : { fixture, stage, runs: [samples(10), samples(11), samples(12)] } + ) + }); + + expect(() => assertTimeToAnswerPromotion(artifact(), slowFirst)).not.toThrow(); + + // Two slow runs and one fast run: the median is slow, so it must fail. + const slowMajority = candidate({ + records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => + fixture === "reference-250" && stage === "commandQueryMs" + ? { + fixture, + stage, + runs: [ + [...Array(15).fill(10)], + [...Array(15).fill(500)], + [...Array(15).fill(500)] + ] + } + : { fixture, stage, runs: [samples(10), samples(11), samples(12)] } + ) + }); + expect(() => assertTimeToAnswerPromotion(artifact(), slowMajority)).toThrow("promotion limit"); + }); + + it("uses phase-normalized p95 as the controlled stage-5 promotion authority", () => { + const stageFiveRuns = (repeatedValue: number) => + Array.from({ length: TIME_TO_ANSWER_CONTROLLED_RUNS }, () => + Array.from({ length: TIME_TO_ANSWER_WARMED_SAMPLES }, (_, index) => (index < 10 ? 10 : repeatedValue)) + ); + const baseline = artifact({ + records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => + fixture === "reference-25" && stage === "publicationToNodeObservationMs" + ? { fixture, stage, runs: stageFiveRuns(10) } + : { fixture, stage, runs: [samples(10), samples(11), samples(12)] } + ) + }); + const candidateStageFiveSlow = candidate({ + records: TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => + fixture === "reference-25" && stage === "publicationToNodeObservationMs" + ? { fixture, stage, runs: stageFiveRuns(100) } + : { fixture, stage, runs: [samples(10), samples(11), samples(12)] } + ) + }); + // Ordinary per-run p95 is 100 in both artifacts, but all-repeat phase medians + // make the normalized figure rise from 10 to 55 and reject promotion. + expect(() => assertTimeToAnswerPromotion(baseline, candidateStageFiveSlow)).toThrow("promotion limit"); + }); + + it("cannot let a cold first observation enter a warmed stage summary", () => { + // The daemon's first passes are cold, and with 15 recorded samples nearest-rank + // p95 IS the maximum — so a surviving cold observation would become the + // published number. The protocol discards a FIXED 60 observations (declared + // before the capture), keeps them all for audit, and never trims further. + const cold = Array.from({ length: TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS }, (_, index) => 99.9 - index); + const warm = Array.from({ length: TIME_TO_ANSWER_WARMED_SAMPLES }, (_, index) => 2 + index * 0.1); + const { warmUps, recorded } = splitWarmedObservations([...cold, ...warm]); + expect(warmUps).toEqual(cold); + expect(warmUps).toHaveLength(cold.length); + expect(recorded).toEqual(warm); + const summary = summarizeTimeToAnswerStage([recorded, recorded, recorded]); + expect(summary.runP95Ms.every((value) => value < 10)).toBe(true); + expect(summary.medianOfThreeRunP95Ms).toBeLessThan(10); + // No arbitrary sampling: the whole window is required and a short window FAILS + // rather than being silently trimmed to the declared count. + expect(() => splitWarmedObservations([...cold, ...warm].slice(0, cold.length + warm.length - 1))).toThrow(); + }); + + it("never adapts the measured window to diagnostics", () => { + const observations = Array.from({ length: 75 }, (_, index) => index < 60 ? 100 - index : 2); + const { warmUps, recorded } = splitWarmedObservations(observations); + expect(warmUps).toHaveLength(60); + expect(recorded).toEqual(Array(15).fill(2)); + }); + + it("binds the executed daemon binary to the recorded revision", () => { + const digest = "a".repeat(64); + expect(() => + assertDaemonBinaryProvenance({ expectedSha256: digest, observedSha256: digest, phase: "before" }) + ).not.toThrow(); + expect(() => + assertDaemonBinaryProvenance({ + expectedSha256: digest, + observedSha256: "b".repeat(64), + phase: "before capture" + }) + ).toThrow("daemon binary provenance failed"); + expect(() => + assertDaemonBinaryProvenance({ expectedSha256: "not-a-digest", observedSha256: digest, phase: "before" }) + ).toThrow("two lowercase sha256 digests"); + }); + + it("classifies every stage as cold-start, warmed-repeated or scenario-specific", () => { + for (const stage of TIME_TO_ANSWER_STAGES) { + expect(TIME_TO_ANSWER_STAGE_KIND[stage.id]).toBeDefined(); + } + // Process start is genuinely cold: its first observation IS the measurement. + expect(TIME_TO_ANSWER_STAGE_KIND.daemonStartToListenerMs).toBe("cold-start"); + expect(TIME_TO_ANSWER_STAGE_KIND.listenerToFirstDockerModelMs).toBe("cold-start"); + // The daemon-side attribution stages are warmed repeated operations. + expect(TIME_TO_ANSWER_STAGE_KIND.dockerObservationMs).toBe("warmed-repeated"); + expect(TIME_TO_ANSWER_STAGE_KIND.composeEnrichmentMs).toBe("warmed-repeated"); + expect(TIME_TO_ANSWER_STAGE_KIND.findingsDerivationMs).toBe("warmed-repeated"); + // Scenario cells are declared only for scenario fixtures. + expect(isScenarioCell("slow-bounded-compose-projection", "composeEnrichmentMs")).toBe(true); + expect(isScenarioCell("reference-25", "composeEnrichmentMs")).toBe(false); + }); + + it("rejects a candidate with different browser flags", () => { + expect(() => + assertTimeToAnswerPromotion( + artifact(), + candidate({ environment: { browserFlags: ["--disable-background-networking", "--enable-gpu"] } }) + ) + ).toThrow("does not match the pinned baseline environment"); + }); + + it("rejects a candidate built in a non-production mode", () => { + expect(() => validateTimeToAnswerEvidence(candidate({ environment: { buildMode: "development" } }))).toThrow( + "closed safe metadata fields" + ); + }); + + it("rejects evidence with a missing stage", () => { + const records = TIME_TO_ANSWER_MATRIX.slice(1).map(({ fixture, stage }) => ({ + fixture, + stage, + runs: [samples(10), samples(11), samples(12)] + })); + expect(() => validateTimeToAnswerEvidence(artifact({ records }))).toThrow("exact fixture × stage matrix"); + }); + + it("rejects evidence with an undeclared stage", () => { + const records = [ + ...TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => ({ + fixture, + stage, + runs: [samples(10), samples(11), samples(12)] + })), + { fixture: "reference-25", stage: "inventedStageMs", runs: [samples(1), samples(1), samples(1)] } + ]; + expect(() => validateTimeToAnswerEvidence(artifact({ records }))).toThrow("exact fixture × stage matrix"); + }); + + it("rejects a malformed sample count", () => { + const records = TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => + fixture === "reference-25" && stage === "commandQueryMs" + ? { fixture, stage, runs: [samples(10).slice(0, 14), samples(11), samples(12)] } + : { fixture, stage, runs: [samples(10), samples(11), samples(12)] } + ); + expect(() => validateTimeToAnswerEvidence(artifact({ records }))).toThrow( + `requires exactly ${TIME_TO_ANSWER_WARMED_SAMPLES} finite` + ); + }); + + it("rejects a stage with too few controlled runs", () => { + const records = TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => + fixture === "reference-100" && stage === "buildModelMs" + ? { fixture, stage, runs: [samples(10), samples(11)] } + : { fixture, stage, runs: [samples(10), samples(11), samples(12)] } + ); + expect(() => validateTimeToAnswerEvidence(artifact({ records }))).toThrow( + "requires exactly three raw runs per stage" + ); + expect(TIME_TO_ANSWER_CONTROLLED_RUNS).toBe(3); + }); + + it("rejects a supplied or fabricated summary instead of recomputing it", () => { + const records = TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => + fixture === "reference-250" && stage === "composeEnrichmentMs" + ? { + fixture, + stage, + runs: [samples(10), samples(11), samples(12)], + summary: { runP95Ms: [1, 1, 1], medianOfThreeRunP95Ms: 1 } + } + : { fixture, stage, runs: [samples(10), samples(11), samples(12)] } + ); + expect(() => validateTimeToAnswerEvidence(artifact({ records }))).toThrow("unsafe or incomplete shape"); + }); + + it("rejects a negative or non-finite sample", () => { + for (const bad of [-1, Number.NaN, Number.POSITIVE_INFINITY, "12" as unknown as number]) { + const records = TIME_TO_ANSWER_MATRIX.map(({ fixture, stage }) => + fixture === "reference-25" && stage === "findingsDerivationMs" + ? { fixture, stage, runs: [[bad, ...samples(10).slice(1)], samples(11), samples(12)] } + : { fixture, stage, runs: [samples(10), samples(11), samples(12)] } + ); + expect(() => validateTimeToAnswerEvidence(artifact({ records }))).toThrow(); + } + }); + + it("rejects an unknown baseline identifier", () => { + expect(() => validateTimeToAnswerEvidence({ ...artifact(), baseline: "dockermap-v1/other" })).toThrow( + "closed baseline/environment/records schema" + ); + }); +}); diff --git a/apps/web/vite.config.ts b/apps/web/vite.config.ts index 0ea2ae44..6ed5fdae 100644 --- a/apps/web/vite.config.ts +++ b/apps/web/vite.config.ts @@ -3,6 +3,16 @@ import react from "@vitejs/plugin-react"; export default defineConfig({ plugins: [react()], + /** + * The benchmark instrumentation seam (#335) is compile-time gated. The + * production build defines the flag `false`, so the acceptance hook, its event + * identifiers, the probe entry and the artificial render-delay machinery are + * all removed by dead-code elimination before minification. Flipping this value + * here is exactly what must never happen: tests/perf/productionIsolation.test.mjs + * reads this file and requires the production definition to be `false`, and + * inspects the built artifact for the seam's identifiers. + */ + define: { __DOCKERMAP_BENCH_ACCEPTANCE__: "false" }, server: { port: 3233 } diff --git a/crates/dockermap-daemon/src/bench_timing.rs b/crates/dockermap-daemon/src/bench_timing.rs new file mode 100644 index 00000000..05152440 --- /dev/null +++ b/crates/dockermap-daemon/src/bench_timing.rs @@ -0,0 +1,156 @@ +//! Test-only stage attribution for the time-to-answer benchmark (#335). +//! +//! This module measures the *current* implementation without changing it. The +//! Docker observation and the Compose filesystem projection are timed as two +//! separately attributable stages even though they still execute inside one +//! publication budget; #336 owns moving the projection off that critical path. +//! +//! It is inert unless `DOCKERMAP_BENCH_STAGE_TIMING_PATH` names an absolute +//! path. When unset or unusable, nothing is measured, nothing is written, and +//! no public response changes. There is no route, no response field, and no +//! production telemetry: the sink is an append-only newline-delimited JSON file +//! chosen by the benchmark harness. + +use std::io::Write; +use std::path::{Path, PathBuf}; +use std::time::Instant; + +const BENCH_STAGE_TIMING_PATH_ENV: &str = "DOCKERMAP_BENCH_STAGE_TIMING_PATH"; + +/// Docker inventory observation (containers, networks, volumes). +pub(crate) const STAGE_DOCKER_OBSERVATION: &str = "dockerObservationMs"; +/// Compose filesystem projection for the same publication. +pub(crate) const STAGE_COMPOSE_ENRICHMENT: &str = "composeEnrichmentMs"; +/// Findings derivation for the published runtime map. +pub(crate) const STAGE_FINDINGS_DERIVATION: &str = "findingsDerivationMs"; + +/// Resolve the configured sink. Only an absolute path is accepted, so a +/// relative value cannot silently land inside a working directory, and an +/// empty or malformed value disables the hook instead of failing a publication. +pub(crate) fn sink_from_env_value(value: Option) -> Option { + let value = value?; + let trimmed = value.trim(); + if trimmed.is_empty() { + return None; + } + let path = Path::new(trimmed); + if !path.is_absolute() { + return None; + } + Some(path.to_path_buf()) +} + +pub(crate) fn sink() -> Option { + sink_from_env_value(std::env::var(BENCH_STAGE_TIMING_PATH_ENV).ok()) +} + +/// One NDJSON record. Kept pure so its shape is testable without touching the +/// process environment or the filesystem. +pub(crate) fn stage_timing_line(stage: &str, milliseconds: f64) -> String { + format!("{{\"stage\":\"{stage}\",\"ms\":{milliseconds:.3}}}\n") +} + +fn write_line(sink: Option<&Path>, line: &str) { + let Some(path) = sink else { + return; + }; + // A benchmark sink must never be able to interrupt collection: a write + // failure is dropped, not propagated. + if let Ok(mut file) = std::fs::OpenOptions::new() + .create(true) + .append(true) + .open(path) + { + let _ = file.write_all(line.as_bytes()); + } +} + +/// Record one stage duration. A no-op when the hook is disabled. +pub(crate) fn record(sink: Option<&Path>, stage: &str, start: Instant) { + if sink.is_none() { + return; + } + let elapsed = start.elapsed(); + write_line( + sink, + &stage_timing_line(stage, elapsed.as_secs_f64() * 1000.0), + ); +} + +#[cfg(test)] +mod tests { + use super::*; + use std::time::Duration; + + #[test] + fn sink_requires_an_absolute_non_empty_path() { + assert!(sink_from_env_value(None).is_none()); + assert!(sink_from_env_value(Some(String::new())).is_none()); + assert!(sink_from_env_value(Some(" ".into())).is_none()); + assert!(sink_from_env_value(Some("relative/bench.jsonl".into())).is_none()); + assert!(sink_from_env_value(Some("./bench.jsonl".into())).is_none()); + assert_eq!( + sink_from_env_value(Some("/tmp/dockermap-bench.jsonl".into())).as_deref(), + Some(Path::new("/tmp/dockermap-bench.jsonl")) + ); + assert_eq!( + sink_from_env_value(Some(" /tmp/dockermap-bench.jsonl ".into())).as_deref(), + Some(Path::new("/tmp/dockermap-bench.jsonl")) + ); + } + + #[test] + fn stage_lines_are_closed_newline_delimited_json() { + assert_eq!( + stage_timing_line(STAGE_DOCKER_OBSERVATION, 12.3456), + "{\"stage\":\"dockerObservationMs\",\"ms\":12.346}\n" + ); + assert_eq!( + stage_timing_line(STAGE_COMPOSE_ENRICHMENT, 0.0), + "{\"stage\":\"composeEnrichmentMs\",\"ms\":0.000}\n" + ); + assert!(stage_timing_line(STAGE_DOCKER_OBSERVATION, 1.0).ends_with('\n')); + } + + #[test] + fn disabled_hook_writes_nothing() { + let directory = tempfile::tempdir().expect("temporary bench directory"); + let target = directory.path().join("bench.jsonl"); + record(None, STAGE_DOCKER_OBSERVATION, Instant::now()); + for _ in 0..3 { + record(None, STAGE_COMPOSE_ENRICHMENT, Instant::now()); + } + assert!( + !target.exists(), + "a disabled bench hook must not create a sink file" + ); + assert!(std::fs::read_dir(directory.path()) + .expect("bench directory") + .next() + .is_none()); + } + + #[test] + fn enabled_hook_appends_one_line_per_stage() { + let directory = tempfile::tempdir().expect("temporary bench directory"); + let target = directory.path().join("bench.jsonl"); + let start = Instant::now(); + std::thread::sleep(Duration::from_millis(1)); + record(Some(&target), STAGE_DOCKER_OBSERVATION, start); + record(Some(&target), STAGE_COMPOSE_ENRICHMENT, start); + let written = std::fs::read_to_string(&target).expect("bench sink is readable"); + let lines = written.lines().collect::>(); + assert_eq!(lines.len(), 2); + assert!(lines[0].starts_with("{\"stage\":\"dockerObservationMs\",\"ms\":")); + assert!(lines[1].starts_with("{\"stage\":\"composeEnrichmentMs\",\"ms\":")); + for line in lines { + assert!(line.ends_with('}')); + let value: serde_json::Value = serde_json::from_str(line).expect("valid JSON line"); + let ms = value + .get("ms") + .and_then(|ms| ms.as_f64()) + .expect("numeric ms"); + assert!(ms.is_finite() && ms >= 0.0); + } + } +} diff --git a/crates/dockermap-daemon/src/cache_refresh.rs b/crates/dockermap-daemon/src/cache_refresh.rs index 7970a919..df3422a3 100644 --- a/crates/dockermap-daemon/src/cache_refresh.rs +++ b/crates/dockermap-daemon/src/cache_refresh.rs @@ -638,6 +638,8 @@ impl DaemonCache { .assign(&mut self.snapshot, &mut self.health, &mut self.runtime_map); // Findings are a pure projection of the sanitized runtime map, so // calculate and cache them only after the publication revision exists. + let bench_sink = crate::bench_timing::sink(); + let findings_started = std::time::Instant::now(); let mut findings = derive_findings(&self.runtime_map); if self.health.mode == RuntimeMode::Docker { if let Some((scan, binding)) = &self.compose_runtime_binding { @@ -647,6 +649,11 @@ impl DaemonCache { } } findings.sort_by(|left, right| left.id.cmp(&right.id)); + crate::bench_timing::record( + bench_sink.as_deref(), + crate::bench_timing::STAGE_FINDINGS_DERIVATION, + findings_started, + ); self.findings = FindingsResponse { summary: FindingSummary::from_findings(&findings), findings, @@ -863,12 +870,22 @@ where + 'static, { let started = tokio::time::Instant::now(); + // Test-only stage attribution (#335). Disabled unless the benchmark harness + // sets an absolute DOCKERMAP_BENCH_STAGE_TIMING_PATH; it records durations + // only and never changes what is collected or published. + let bench_sink = crate::bench_timing::sink(); + let bench_docker_started = std::time::Instant::now(); let observation = match tokio::time::timeout(snapshot_timeout, collector.collect_observation()).await { Ok(Ok(observation)) => observation, Ok(Err(error)) => return Err(DockerReadFailure::Failed(error)), Err(_) => return Err(DockerReadFailure::TimedOut), }; + crate::bench_timing::record( + bench_sink.as_deref(), + crate::bench_timing::STAGE_DOCKER_OBSERVATION, + bench_docker_started, + ); let mut snapshot = observation.snapshot; snapshot.images = derive_images(&snapshot); let collected_at = snapshot.last_updated; @@ -885,10 +902,21 @@ where let _flight = flight; projection(observation.compose_containers, collected_at) }); - match tokio::time::timeout(remaining, projection_task).await { + let projection_started = std::time::Instant::now(); + let binding = match tokio::time::timeout(remaining, projection_task).await { Ok(Ok(binding)) => binding, Ok(Err(_)) | Err(_) => None, - } + }; + // Attributed separately from the Docker observation so the + // baseline can show that this projection currently sits inside + // the Docker publication budget. #336 owns moving it off that + // path; nothing is decoupled here. + crate::bench_timing::record( + bench_sink.as_deref(), + crate::bench_timing::STAGE_COMPOSE_ENRICHMENT, + projection_started, + ); + binding } else { None } diff --git a/crates/dockermap-daemon/src/main.rs b/crates/dockermap-daemon/src/main.rs index 7d0b5e52..a309054c 100644 --- a/crates/dockermap-daemon/src/main.rs +++ b/crates/dockermap-daemon/src/main.rs @@ -1,4 +1,5 @@ mod auth; +mod bench_timing; mod cache_refresh; mod compose_api; mod config; diff --git a/docs/testing/TIME_TO_ANSWER_BASELINE.md b/docs/testing/TIME_TO_ANSWER_BASELINE.md new file mode 100644 index 00000000..cb793652 --- /dev/null +++ b/docs/testing/TIME_TO_ANSWER_BASELINE.md @@ -0,0 +1,202 @@ +# Time-to-answer Baseline 4 + +Status: **accepted candidate under review in PR #346**. Baseline 4 is the intended +comparison authority for #336 and #337 once that PR is merged. It is not represented +as merged or as an unconditional authority before review completes. + +- baseline id / methodology: `dockermap-v1/time-to-answer-methodology-8` +- product and capture-harness revision: `bdce6ae354d757d3318c514e10d620edb497918e` +- durable evidence root: `/srv/jonas/evidence/dockermap/time-to-answer/bdce6ae/` + (stored outside this repository under the evidence-artifact policy) +- general raw artifact: `time-to-answer-baseline-4.json`, sha256 + `916610bdd4767fb01a4f57e29f318aa504836875cbb4d87ba615c70cb6fb7300` +- capture duration: 105.1 minutes; 44 records × 3 runs × 15 measured samples = + 1,980 raw samples + +Baselines 1, 2, and 3 are **REJECTED** historical attempts and are not current +figures or authority for any comparison. Baseline 1 used an uncommitted harness and +incorrect stage boundaries; Baseline 2 retained cold data and shared a stage clock; +Baseline 3 did not sweep poll phase, retained an insufficient warm-up, and had +incorrect provenance/compatibility handling. Their measurement numbers do not appear +in this record. + +## Conditioning and composite contract + +Every ordinary warmed end-to-end cell uses exactly 60 fixed burn-in observations +followed by exactly 15 measured observations. Observation 61 is always the first +measured sample. Burn-in is retained for audit but excluded completely from timing +summaries and promotion comparisons. This is fixed, deterministic conditioning for +baseline and candidate; it makes **no stationarity claim**. + +The composite authority has 44 rows: 38 `end-to-end` rows from +`time-to-answer-baseline-4.json`, plus 6 `controlled-poll-phase` rows from +`time-to-answer-stage5.json`. Every row records fixture, stage, measurement protocol, +source evidence file, and the checkpoint above. All 44 composite runs equal their raw +source runs; there are no duplicate cells. + +Normal Stage-6 and Stage-7 timing rows are ordinary `end-to-end` measurements. +The separate `controlled-stage6-stage7-seam-isolation` control is supporting evidence +only and is **FAIL** at this checkpoint: + +``` +[independence] FAIL: Error: control delay did not begin after the acceptance timestamp +``` + +Its disclosed limitation is `publication-level causal identity unavailable` and +`validatesDaemonToBrowserAttribution: false`. It is not a timing row, its samples are +not Baseline-4 timing observations, and it is not daemon→browser attribution evidence. + +## Recomputed timing table + +This table is copied from the recomputed summary +`time-to-answer-baseline-4-summary.md`, derived from the composite raw sample arrays. + +| fixture | stage | run p95 (ms) | reviewed aggregation | reviewed (ms) | min | max | +| --- | --- | --- | --- | --- | --- | --- | +| reference-25 | daemonStartToListenerMs | 38.22 / 37.28 / 41.97 | median-of-three-run-p95 | 38.22 | 33.77 | 41.97 | +| reference-25 | listenerToFirstDockerModelMs | 11.86 / 11.07 / 11.50 | median-of-three-run-p95 | 11.50 | 9.13 | 11.86 | +| reference-25 | dockerObservationMs | 1.74 / 1.71 / 1.84 | median-of-three-run-p95 | 1.74 | 1.19 | 1.84 | +| reference-25 | composeEnrichmentMs | 1.01 / 1.06 / 1.02 | median-of-three-run-p95 | 1.02 | 0.62 | 1.06 | +| reference-25 | notificationToCoherentModelMs | 21.20 / 22.40 / 18.00 | median-of-three-run-p95 | 21.20 | 11.20 | 22.40 | +| reference-25 | coherentModelToUsefulRenderMs | 51.00 / 37.60 / 30.20 | median-of-three-run-p95 | 37.60 | 17.60 | 51.00 | +| reference-25 | buildModelMs | 0.20 / 0.20 / 0.20 | median-of-three-run-p95 | 0.20 | 0.00 | 0.20 | +| reference-25 | findingsDerivationMs | 0.08 / 0.06 / 0.08 | median-of-three-run-p95 | 0.08 | 0.04 | 0.08 | +| reference-25 | legacyTopologyLayoutMs | 1.90 / 1.90 / 1.90 | median-of-three-run-p95 | 1.90 | 1.60 | 1.90 | +| reference-25 | commandQueryMs | 52.60 / 55.50 / 59.90 | median-of-three-run-p95 | 55.50 | 6.00 | 59.90 | +| reference-25 | productionBundleMs | 50.30 / 54.70 / 58.60 | median-of-three-run-p95 | 54.70 | 37.80 | 58.60 | +| reference-25 | publicationToNodeObservationMs | 1897.38 / 1898.20 / 1898.67 | phase-normalized-p95 | 1897.79 | 97.18 | 1898.67 | +| reference-100 | daemonStartToListenerMs | 118.14 / 129.97 / 110.31 | median-of-three-run-p95 | 118.14 | 99.22 | 129.97 | +| reference-100 | listenerToFirstDockerModelMs | 22.30 / 18.32 / 19.11 | median-of-three-run-p95 | 19.11 | 15.23 | 22.30 | +| reference-100 | dockerObservationMs | 3.24 / 4.34 / 2.79 | median-of-three-run-p95 | 3.24 | 2.16 | 4.34 | +| reference-100 | composeEnrichmentMs | 1.07 / 1.01 / 1.05 | median-of-three-run-p95 | 1.05 | 0.63 | 1.07 | +| reference-100 | notificationToCoherentModelMs | 35.90 / 37.20 / 45.80 | median-of-three-run-p95 | 37.20 | 29.10 | 45.80 | +| reference-100 | coherentModelToUsefulRenderMs | 52.40 / 53.80 / 48.20 | median-of-three-run-p95 | 52.40 | 35.10 | 53.80 | +| reference-100 | buildModelMs | 0.50 / 0.50 / 0.40 | median-of-three-run-p95 | 0.50 | 0.20 | 0.50 | +| reference-100 | findingsDerivationMs | 0.21 / 0.23 / 0.26 | median-of-three-run-p95 | 0.23 | 0.16 | 0.26 | +| reference-100 | legacyTopologyLayoutMs | 28.30 / 23.10 / 24.40 | median-of-three-run-p95 | 24.40 | 19.40 | 28.30 | +| reference-100 | commandQueryMs | 18.20 / 40.10 / 22.80 | median-of-three-run-p95 | 22.80 | 8.20 | 40.10 | +| reference-100 | productionBundleMs | 52.00 / 50.50 / 61.10 | median-of-three-run-p95 | 52.00 | 37.20 | 61.10 | +| reference-100 | publicationToNodeObservationMs | 1898.07 / 1898.53 / 1897.64 | phase-normalized-p95 | 1897.61 | 98.18 | 1898.53 | +| reference-250 | daemonStartToListenerMs | 311.26 / 325.69 / 308.36 | median-of-three-run-p95 | 311.26 | 106.85 | 325.69 | +| reference-250 | listenerToFirstDockerModelMs | 170.76 / 123.58 / 121.55 | median-of-three-run-p95 | 123.58 | 42.40 | 170.76 | +| reference-250 | dockerObservationMs | 7.26 / 6.21 / 7.46 | median-of-three-run-p95 | 7.26 | 4.42 | 7.46 | +| reference-250 | composeEnrichmentMs | 0.94 / 0.97 / 1.14 | median-of-three-run-p95 | 0.97 | 0.63 | 1.14 | +| reference-250 | notificationToCoherentModelMs | 91.00 / 91.30 / 90.90 | median-of-three-run-p95 | 91.00 | 70.30 | 91.30 | +| reference-250 | coherentModelToUsefulRenderMs | 188.00 / 191.00 / 188.70 | median-of-three-run-p95 | 188.70 | 145.40 | 191.00 | +| reference-250 | buildModelMs | 1.40 / 1.70 / 1.90 | median-of-three-run-p95 | 1.70 | 0.60 | 1.90 | +| reference-250 | findingsDerivationMs | 0.81 / 1.03 / 0.80 | median-of-three-run-p95 | 0.81 | 0.58 | 1.03 | +| reference-250 | legacyTopologyLayoutMs | 150.40 / 166.00 / 149.10 | median-of-three-run-p95 | 150.40 | 123.30 | 166.00 | +| reference-250 | commandQueryMs | 26.10 / 28.70 / 28.40 | median-of-three-run-p95 | 28.40 | 11.40 | 28.70 | +| reference-250 | productionBundleMs | 52.40 / 60.40 / 54.70 | median-of-three-run-p95 | 54.70 | 40.90 | 60.40 | +| reference-250 | publicationToNodeObservationMs | 1898.24 / 1897.54 / 1898.61 | phase-normalized-p95 | 1897.67 | 96.58 | 1898.61 | +| slow-bounded-compose-projection | composeEnrichmentMs | 6.62 / 9.55 / 11.77 | median-of-three-run-p95 | 9.55 | 5.13 | 11.77 | +| provider-only-revision-change | notificationToCoherentModelMs | 45.50 / 41.30 / 36.60 | median-of-three-run-p95 | 41.30 | 27.80 | 45.50 | +| provider-only-revision-change | publicationToNodeObservationMs | 1896.43 / 1898.67 / 1898.03 | phase-normalized-p95 | 1898.00 | 97.02 | 1898.67 | +| docker-topology-change | notificationToCoherentModelMs | 46.60 / 43.70 / 48.30 | median-of-three-run-p95 | 46.60 | 29.40 | 48.30 | +| docker-topology-change | coherentModelToUsefulRenderMs | 46.70 / 52.40 / 44.50 | median-of-three-run-p95 | 46.70 | 37.30 | 52.40 | +| docker-topology-change | publicationToNodeObservationMs | 1898.41 / 1897.48 / 1898.77 | phase-normalized-p95 | 1897.88 | 97.46 | 1898.77 | +| unavailable-optional-provider | notificationToCoherentModelMs | 44.30 / 50.10 / 48.50 | median-of-three-run-p95 | 48.50 | 26.50 | 50.10 | +| unavailable-optional-provider | publicationToNodeObservationMs | 1898.45 / 1898.15 / 1897.72 | phase-normalized-p95 | 1897.93 | 97.40 | 1898.45 | + +## Stage 5 controlled poll-phase results + +The dedicated `controlled-poll-phase` protocol owns arm → mark → trigger → +identity-acknowledgement. It covers all six declared +`publicationToNodeObservationMs` fixtures on ten declared phases (100 through 1900 +ms) over the 2000 ms poll interval. There are 45 distinct trigger ids per fixture; +the maximum absolute observed-vs-declared phase error is at most 1.0 ms. The published +figure is phase-normalized, not user-traffic or network latency. + +The recomputed phase table prints these per-phase medians for the reference and +topology fixtures: + +| fixture | declared phases | observed latency median per declared phase (ms, earliest→latest) | phase-normalized p95 (ms) | span (ms) | +| --- | --- | --- | --- | --- | +| reference-25 | 10 | 1898, 1699, 1498, 1299, 1098, 899, 698, 498, 298, 97 | 1897.79 | 1801.49 | +| reference-100 | 10 | 1898, 1698, 1498, 1298, 1098, 898, 698, 498, 298, 99 | 1897.61 | 1800.34 | +| reference-250 | 10 | 1898, 1698, 1498, 1298, 1098, 898, 698, 499, 299, 97 | 1897.67 | 1802.03 | +| docker-topology-change | 10 | 1898, 1698, 1498, 1298, 1098, 899, 698, 498, 298, 99 | 1897.88 | 1801.32 | + +The phase-normalized figure weights declared phases uniformly. The complete six-row +reviewed values are in the timing table above; the summary's dedicated phase table is +the source for the printed per-phase curves. + +## Bucket shares + +Source: `time-to-answer-baseline-4-summary.md`. + +### reference-25 buckets (sum of reviewed stage figures: 2121.45 ms) +- transport-notification: 1897.79 ms (89.5%) +- rendering: 94.20 ms (4.4%) +- search: 55.50 ms (2.6%) +- backend-collection: 52.56 ms (2.5%) +- browser-model: 21.40 ms (1.0%) + +### reference-100 buckets (sum of reviewed stage figures: 2228.69 ms) +- transport-notification: 1897.61 ms (85.1%) +- backend-collection: 141.78 ms (6.4%) +- rendering: 128.80 ms (5.8%) +- browser-model: 37.70 ms (1.7%) +- search: 22.80 ms (1.0%) + +### reference-250 buckets (sum of reviewed stage figures: 2856.44 ms) +- transport-notification: 1897.67 ms (66.4%) +- backend-collection: 443.88 ms (15.5%) +- rendering: 393.80 ms (13.8%) +- browser-model: 92.70 ms (3.2%) +- search: 28.40 ms (1.0%) + +## Burn-in audit + +`time-to-answer-baseline-4.json.harness-evidence.json` proves 96 warmed cells with +complete 75-observation windows: burn-in is observations 1–60, measurement is +observations 61–75, and there are zero mismatches. Eighteen cold-start cells have no +burn-in. The retained burn-in data is audit material only and never contributes to a +timing summary. + +## Pinned environment and provenance + +| field | value | +| --- | --- | +| runnerClass | linux-x86_64-dedicated | +| cpuClass | cpus-16vcpu | +| osImage / kernel | ubuntu-26.04 / 7.0.0-31-generic | +| Node / Rust / Docker | 22.23.2 / 1.88.0 / 29.8.1 | +| SSE poll interval | 2000 ms | +| sourceRevision / harnessRevision | `bdce6ae354d757d3318c514e10d620edb497918e` / `bdce6ae354d757d3318c514e10d620edb497918e` | +| daemon binary SHA-256 | `862a70cac056dcdbfc0593a03050adabc14a5f7a780c3e64873bec905f772807` | +| daemon build command | `cargo build --release --locked -p dockermap-daemon --manifest-path crates/Cargo.toml` | +| cargo revision | cargo-1.88.0-873a06493-2025-05-10 | +| browser engine / revision | chromium / 1.61.0 | +| browser flags | `--disable-background-networking --disable-sync --no-first-run --no-default-browser-check` | +| font environment / build mode | system-default / production | +| fixture revision | dockermap-v1/time-to-answer-fixtures-1 | +| methodology version | dockermap-v1/time-to-answer-methodology-8 | + +The daemon digest was identical before and after capture. The raw sections are +assembled with: + +``` +npx tsx tests/perf/assembleCompositeEvidence.ts --general --stageFive --output +``` + +Recompute the published summary with: + +``` +npm run perf:summarize -- --artifact +``` + +The artifact paths are external evidence-artifact-policy storage, not repository +deliverables. + +## What this baseline does NOT claim + +- It is not real-Docker latency: it uses a deterministic fixture daemon. +- It is not a real network or user-traffic distribution. +- Stage-5 phase-normalized timing is not network latency and does not claim + publications occur uniformly across poll phase. +- It does not prove the model is complete: Stages 1/2 and Stage 6 end at coherence. +- It makes no daemon-publication attribution claim from the failed seam-isolation + control. +- It grants no permission to optimise. #336, #337, and #338 must compare in a + compatible pinned environment under `max(baseline × 1.25, baseline + 2 ms)`. diff --git a/docs/testing/TIME_TO_ANSWER_EVIDENCE.md b/docs/testing/TIME_TO_ANSWER_EVIDENCE.md new file mode 100644 index 00000000..69fcb305 --- /dev/null +++ b/docs/testing/TIME_TO_ANSWER_EVIDENCE.md @@ -0,0 +1,699 @@ +# Time-to-answer evidence + +## Methodology-8 Baseline-4 conditioning (authoritative) + +For every ordinary warmed end-to-end fixture/run, Baseline-4 performs **exactly +60 fixed burn-in observations**, retains all 60 in harness evidence, then records +**exactly 15 measured observations**. Observation **61** is always the first +measured sample. Burn-in is excluded completely from timing summaries and +promotion comparisons. It is a deterministic, equal conditioning workload for a +baseline and candidate; it makes no claim that 60 guarantees steady state. + +No observed value may infer stationarity, adaptively trim samples, select a +per-metric warm-up, or extend the burn-in. Stationarity/drift calculations remain +historical informational diagnostics only and never alter or invalidate an +otherwise structurally valid ordinary run. + +The final 60-observation calibration is retained as **REJECTED, +NON-AUTHORITATIVE** audit evidence for Baseline-4. It disproved a common sustained +stationarity validity rule: several metrics stabilized in 2–3 observations, +`findingsDerivationMs` near 13, `legacyTopologyLayoutMs` near 26, +`commandQueryMs` near 43, and `buildModelMs` never met the criterion within 60. +Consequently Baseline-4 does not require calibration PASS, frozen per-metric +counts, the former +2 margin, or the 0.5–1.5x band. + +Controlled evidence is distinct: Stage 5 is `controlled-poll-phase`; Stage-6/7 +seam isolation is `controlled-stage6-stage7-seam-isolation`. Neither contributes +artificial samples to the normal end-to-end dataset. The normal Stage-6 and +Stage-7 timings remain ordinary end-to-end 60+15 measurements. A composite is +incomplete when required Stage-5 controlled evidence is missing; seam-isolation +evidence is supporting validation rather than a prerequisite for normal timing. + +Status: measurement authority for issue #335 and its parent epic #333. This is +**not** an optimization, a product claim, or permission to cut features for a +number. Nothing here changes what DockerMap collects or publishes. + +DockerMap had controlled performance evidence for the Atlas route only. That is +not evidence about the operator path, which starts when the daemon process +starts and ends when a human can act on an answer. This document defines how +that path is measured, what each number proves, and what it deliberately does +not. + +## The contract lives in code, not in this document + +`apps/web/src/lib/performance/timeToAnswerEvidence.ts` is the closed schema and +the math. It contains no timings. It defines: + +- the **12 measured stages**, each with its bucket, a `measures` sentence and a + `doesNotProve` sentence; +- the **fixtures**: `reference-25`, `reference-100`, `reference-250`, plus the + scenario fixtures `provider-only-revision-change`, `docker-topology-change`, + `slow-bounded-compose-projection` and `unavailable-optional-provider`; +- the **exact fixture × stage matrix** — **44 cells** — derived from each stage's + fixture list, never hand-listed; +- the **pinned environment allowlist** (runner class, CPU class, OS image and + kernel, Node/Rust/Docker revisions, Chromium revision and flags, font + environment, production build mode, fixture revision, source revision); +- **raw-sample validation**: 15 warmed samples in each of 3 complete controlled + runs, nearest-rank p95 per run, median of the three run p95 values except that + controlled stage 5 is reviewed and promoted by its phase-normalized p95; +- the **stage-6/7 seam-isolation control** (`assertStageSixSevenIndependence`): a + positive artificial presentation delay injected *after* coherent-model + acceptance must move stage 7 by at least 70% of that delay and must not move + stage 6 beyond `max(30 ms, 25%)`; +- the **promotion gate** `max(baseline × 1.25, baseline + 2 ms)`, compared only + between environments that match on every pinned field except `sourceRevision` + (which differs by design) and `dockerRevision` (recorded but informational: no + measured stage exercises the host Docker daemon). The other 15 fields must + match. + +Because summaries are recomputed from the raw samples at review time, a supplied +summary cannot influence a result. An artifact with a fabricated summary field, +a truncated matrix, a wrong sample count, a negative or non-numeric sample, an +unknown stage, a duplicate record, an undeclared fixture, or an extra/unsafe +metadata field is rejected — see `timeToAnswerEvidence.test.ts`. + +## Stage buckets + +| bucket | stages | what it answers | +| --- | --- | --- | +| backend-collection | daemonStartToListenerMs, listenerToFirstDockerModelMs, dockerObservationMs, composeEnrichmentMs | how long DockerMap takes to have an authoritative answer | +| transport-notification | publicationToNodeObservationMs | how long a published revision takes to become visible | +| browser-model | notificationToCoherentModelMs, buildModelMs | how long the browser needs to turn it into a model | +| rendering | coherentModelToUsefulRenderMs, legacyTopologyLayoutMs, productionBundleMs | how long the operator waits for something useful on screen | +| search | commandQueryMs | how long a direct question takes to answer | + +The buckets exist so the baseline can say **where** the time went — Compose, +notification, model rebuilds, legacy layout or search — instead of only how much +there was. + +## What each number does and does not prove + +The contract carries a `measures`/`doesNotProve` pair for every stage; two +examples of the distinction that matter most: + +- `listenerToFirstDockerModelMs` measures until the first authoritative Docker + model is observable. It does **not** prove the model is complete — optional + provider evidence may still be missing, and a fast number here must never be + read as "the host is fully described". +- `composeEnrichmentMs` measures Compose filesystem correlation separately from + the Docker observation. While the two remain coupled inside one publication + budget, this stage is **measured, not removed**; #336 owns moving it off the + critical path. +- `dockerObservationMs` is measured against a deterministic local fixture + daemon. It is **not** a claim about a real Docker daemon's latency, host load, + or image size. +- `productionBundleMs` and `commandQueryMs` are measured against the pinned + local production build and a fixed representative query set. They are **not** + claims about a real network or about operator behaviour. + +## Controlled run record + +Run exactly three times on the same dedicated pinned runner after a clean +production build. Record every field of the pinned environment. A missing field, +a changed fixture/browser/policy/font/runner class, or a runner health failure +**invalidates the record**; it does not justify retrying until a preferred +duration appears. + +- Keep **raw timings** in the artifact; derive summaries during review. +- Ordinary unit tests may validate evidence shape and math, and must **not** + pretend to be the controlled benchmark. `npm run test:perf` is shape/math and + CPU-only; it never compares elapsed production time. +- Store only sanitized JSON evidence. Never record live host data, raw model or + evidence values, credentials, container identities or screenshots. +- The baseline artifact is an external reviewed record (the same policy the + Atlas evidence uses): it is passed to the benchmark job, not committed. The + baseline identity (`dockermap-v1/time-to-answer-baseline-1`) and the pinned + environment are what are checked in. + +## Fixture source + +`tests/perf/dockerFixtureTopology.mjs` generates deterministic, secret-free +Docker inventory for a given container count and scenario, and +`tests/perf/fake-docker-api.mjs` serves it over a unix socket through the same +three read-only endpoints the daemon's collector uses +(`/containers/json`, `/networks`, `/volumes`, plus `/_ping`, `/version`, +`/info`). + +The daemon is pointed at that socket with +`DOCKERMAP_DOCKER_GATEWAY_SOCKET=` — the same env var it already +uses for the read-only gateway — so the benchmark exercises the **real** +collector, projection and publication path. It never contacts a real Docker +daemon, and the fixture daemon never reads the host filesystem or the network. +25/100/250 containers is why a deterministic fixture daemon is required at all: +inventing 250 real containers on a shared host would not be reproducible. + +Verified working end to end: the release-built daemon, pointed at the fixture +socket with 25 containers, reports `mode: docker`, `dockerReachable: true`, and +publishes 25 containers / 1 network / 5 volumes with a model revision. + +**Generation delta.** The harness can advance a fixture's topology generation +(`POST /__fixture/topology-generation/`), which changes the container +identities/labels and stops the first `n` containers. Generation 0 is the pristine +all-running inventory for every fixture; generation `n` therefore makes the +product render exactly `n` offline/attention services, which is what the stage-7 +expected-content check asserts (`expectedExitedCount` derives the expectation from +the same generator the fixture daemon serves). Earlier revisions of this harness +gave `docker-topology-change` a fixed one-in-three exited mix; that mix is gone, +because a constant mix cannot discriminate a stale render from a fresh one. + +**Scenario premises are asserted, not named.** The capture fails when +`provider-only-revision-change`'s Docker inventory changes, when the +`unavailable-optional-provider` fixture's optional provider is fresh, and when the +`slow-bounded-compose-projection` project does not actually declare its +`SLOW_COMPOSE_SERVICES = 400` services — an empty or truncated project would +otherwise record a cell and look like a fast projection. The `docker-topology-change` +and reference fixtures need no extra premise: the stage-7 expected-content check +binds them to the generation the harness triggered. + +## Promotion rules + +A candidate passes only when, in an equivalent controlled environment, **every** +fixture × stage value is at most `max(baseline × 1.25, baseline + 2 ms)`. Limits +are derived from the measured baseline, never invented as aspirational absolute +milliseconds. A candidate that fails the environment check fails closed; it is +not "close enough". + +**Provenance is not compatibility.** The environment records 20 pinned fields, and +they are used in three different ways: + +| use | fields | how it is treated | +| --- | --- | --- | +| comparison requirements (16) | `runnerClass`, `cpuClass`, `osImage`, `osKernel`, `nodeRevision`, `rustRevision`, `ssePollIntervalMs`, `daemonBinaryBuild`, `cargoRevision`, `browserEngine`, `browserRevision`, `browserFlags`, `fontEnvironment`, `buildMode`, `fixtureRevision`, `methodologyVersion` | must be identical, or the comparison fails closed | +| provenance/identity (3) | `sourceRevision`, `harnessRevision`, `daemonBinarySha256` | recorded so the artifact identifies exactly what was measured; NEVER required to match | +| informational (1) | `dockerRevision` | recorded because it is part of the runner's identity; no measured stage exercises the host Docker daemon | + +`daemonBinarySha256` in particular must not gate a comparison: the candidate's +daemon is **rebuilt from the candidate checkout**, so any legitimate change under +`crates/` — exactly what #336 does — produces a different digest, and a +byte-identical digest is not reproducible across a changed `CARGO_HOME`. + +No optimization claim in #336/#337/#338 (or later) may be accepted without +comparing against this baseline under this rule. + +## Stage attribution inside the daemon + +`dockerObservationMs`, `composeEnrichmentMs` and `findingsDerivationMs` are +measured by a **test-only** hook in the daemon +(`crates/dockermap-daemon/src/bench_timing.rs`). It is inert unless +`DOCKERMAP_BENCH_STAGE_TIMING_PATH` names an absolute path; it then appends +newline-delimited JSON records to that file. There is no route, no response +field, no runtime telemetry, and no behaviour change. + +The hook times the Docker inventory read and the Compose filesystem projection +**separately while both still execute inside the same Docker publication +budget**. This baseline is therefore expected to show that Compose projection +currently sits inside the Docker critical path. That is the measurement, not a +fix: **nothing is decoupled here, and #336 owns moving the projection off that +path** — these are the numbers it must improve against. + +## Stage 6 and stage 7: two clocks, and why they cannot be one + +The two browser stages answer different questions and must not share a clock. + +- **Stage 6 — `notificationToCoherentModelMs`.** Starts when the real stream + notifies the browser of a new model revision. Ends when the **application + accepts one coherent model** — the instant a fetched snapshot/runtime pair with + matching generation, provenance and non-empty model revision becomes the model + the UI renders. +- **Stage 7 — `coherentModelToUsefulRenderMs`.** Starts at the *stage-6 + timestamp*. Ends when the accepted model's **expected Home content is present**, + in a commit the application stamped with that accepted revision, followed by a + **bounded render/presentation confirmation**: the probe discovers the commit from + an animation-frame loop and then awaits a bounded frame after it, so the + end-of-stage segment is one or two frames rather than a fixed number. No sleeps + are involved, and the raw audit records the commit→end duration of every sample so + the mechanism is checkable rather than asserted. + +Stage 6 is observed at the **real application seam**: the acceptance point is +inside `useSystemModel`, at the moment the composed model is published. It is +never inferred from a DOM mutation — baseline 2 was rejected precisely because +its stage 7 was element-for-element identical to stage 6 in all 180 samples. + +Stage 6 has two measurement modes, and which one applies is decided by the closed +matrix, never by the sample: + +- **content mode** — for every fixture that also declares stage 7: the sample ends + only when the (accepted model, rendered content) pair that carries the expected + Home content for the triggered change is observed, so an intermediate + publication that moves no Home metric cannot be mis-attributed to the sample. +- **acceptance-only mode** — for the two provider-state fixtures, whose published + revision deliberately carries no inventory change: stage 6 ends at the + acceptance instant, and stage 7 is not declared for them (requiring a Home + repaint there would be an empty number). + +**Expected content, not just any repaint.** The fixture's generation delta stops +the first `g` containers, so generation `g` renders exactly `g` offline/attention +services. Stage 7 requires the Home metric region to repaint with that exact value +for the accepted revision, so a stale render, an unrelated repaint (such as the +topbar clock) or a render belonging to a different revision cannot end it. The +probe records the pre-change metric value as it arms, so "the DOM changed" is +measured rather than assumed. + +**The chain of custody is checked.** For every browser sample the harness records +which paired API fetch delivered the accepted revision and which notification +preceded that fetch cycle, so stage 6 starts at a notification that provably caused +the fetch the model came from. It does **not** require the accepted revision to +equal a revision this harness's own stream announced: the daemon is read per +request, so `/daemon/health` (what the stream carries) and `/daemon/snapshot` (what +the accepted pair carries) can hold different revisions while a host is churning, +and every SSE connection polls on its own phase. The overlap with the harness's own +stream is recorded as evidence (`acceptedRevisionInApiStream`), and for every cell +that declares stage 7 the accepted revision is additionally bound to the fixture's +triggered generation by the expected-content check. + +## Benchmark-mode application build and production isolation + +Stage 6 needs a signal that only exists in application code, so the seam is +**real product source** (`apps/web/src/lib/performance/modelAcceptance.tsx`) and +the build decides whether it exists: + +| build | flag | what it contains | +| --- | --- | --- | +| production (`apps/web/vite.config.ts`) | `__DOCKERMAP_BENCH_ACCEPTANCE__ = "false"` | no seam, no event identifier, no probe entry, no delay machinery | +| benchmark mode (`tests/perf/benchAppVite.config.mjs`) | `__DOCKERMAP_BENCH_ACCEPTANCE__ = "true"` | the same real app **with** the acceptance seam | +| benchmark probe (`tests/perf/benchVite.config.mjs`) | — | stages 8 and 10 (real `buildModel`/`layout` modules, in Chromium) | + +The application source is never copied or forked: the same files are built twice. +The product build eliminates every benchmark branch by dead-code elimination +before minification, and that is asserted against the built artifact — the +production bundle must not contain `__dockermapBenchAcceptanceSink`, +`__dockermapBenchRenderDelayMs`, `dockermapAcceptedRevision` or any harness +identifier (`tests/perf/productionIsolation.test.mjs`), and the capture refuses to +run at all unless the benchmark-mode build carries the seam and the production +build does not (`assertBuildIsolation`). The production Vite config must define +the flag as the literal `"false"`, which the same suite checks by reading it. + +Which build serves which stage: stages **6 and 7** are measured on the +benchmark-mode application build, **stage 11 (Cmd-K)** and **stage 12 (production +bundle/startup)** on the ordinary production build, and **stages 8 and 10** on the +benchmark-only module probe. The seam emits only an opaque timestamp plus the +model revision token, into an in-memory page sink, and stamps the same opaque +token on the document root so a DOM repaint can be attributed to a revision. +There is no product payload, no network call, no telemetry and no analytics, and +the seam adds no route, no API field and no public schema. + +### Diagnostics-only layer capture + +The benchmark-mode application also keeps a bounded, in-page diagnostic record +for every coherent model it accepts. The capture harness drains those records to +`layers.jsonl`, alongside `lifecycle.jsonl` for browser, context, page, +navigation, teardown, exception, and closure events. Set +`DOCKERMAP_BENCH_DIAG_DIR` to choose the directory; otherwise they are written +to the capture raw directory (or beside the raw capture output). These JSONL +files are observational only: they are not artifact fields, inputs to timing, +warm-up, stationarity, independence, or promotion rules, and an append failure +cannot change a measurement result. Both the diagnostic identifiers and the +in-page sink are compiled out of the ordinary production web bundle. + +## Stage 6/7 controlled seam isolation + +Seam isolation is its own dedicated supporting protocol (`npm run perf:independence`, +`tests/perf/captureIndependence.ts`) and is not part of the general end-to-end capture. +For each fixture that declares both stages it runs the normal three controlled runs of +15 measured samples, then `TIME_TO_ANSWER_INDEPENDENCE_SAMPLES` (3) control samples in +which `__dockermapBenchRenderDelayMs = TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS` (250 ms) +withholds a *newly accepted* publication from the render tree — an artificial +presentation delay injected **after** acceptance. It creates and tears down its own +private fixture, daemon, API, benchmark build and browser contexts and refuses partial +output. The rule enforced before validation: + +- **stage 6 must not move** by more than `max(30 ms, 25% of its median)`; +- **stage 7 must absorb** at least 70% of the injected delay; +- no control stage-7 sample may be shorter than the injected delay (which would + mean the delay never reached the page). + +The verdict, the per-run sample sets and the per-sample audit trail are written to the +protocol's own `--output` artifact and its `--raw-dir` raw record (the raw record is +retained even when the control fails), and the assembled composite records the verdict in +`.supporting-evidence.json`. The closed Baseline-4 evidence schema is +unchanged: the control never becomes artifact timing content and its samples are never +Baseline-4 timing rows. At this checkpoint the control's status is **FAIL** +(`[independence] FAIL: Error: control delay did not begin after the acceptance +timestamp`). The same rule is unit-tested (`timeToAnswerIndependence.test.ts`), including +the RED cases "the delayed render does not move stage 7" and "stage 6 moves with the +delayed presentation". + +Each control sample arms the browser probe and its one-shot delay before advancing +the fixture generation. The delay is consumed only after the real +`useSystemModel` acceptance timestamp, and the accepted snapshot/runtime-map pair +must have one non-empty matching revision. **Limitation:** the current architecture +cannot reliably observe publication-level causal identity from the benchmark +trigger to that accepted pair. This protocol therefore makes no daemon-publication +attribution claim and does not validate daemon-to-browser attribution; it does not +replace that limitation with timing proximity, sequence proximity, or matching +visible content. + +## Running the benchmark + +``` +# 1. pin the environment from the runner itself +npm run perf:metadata -- --output /tmp/time-to-answer-metadata.json +# 2. end-to-end capture of stages 1-4 and 6-12 (3 controlled runs x 15 measured +# samples per declared cell, after the fixed 60-observation burn-in) +npm run perf:time-to-answer -- \ + --metadata /tmp/time-to-answer-metadata.json \ + --output /tmp/time-to-answer-general.json \ + --raw-dir /tmp/time-to-answer-raw \ + --checkpoint +# 3. dedicated controlled Stage-5 capture; the only owner of the poll-phase protocol +npm run perf:stage-five -- \ + --metadata /tmp/time-to-answer-metadata.json \ + --output /tmp/time-to-answer-stage5.json \ + --raw-dir /tmp/time-to-answer-stage5-raw +# 4. dedicated Stage-6/7 seam-isolation control: supporting evidence only, never a +# Baseline-4 timing row (the companion record states its verdict) +npm run perf:independence -- \ + --metadata /tmp/time-to-answer-metadata.json \ + --output /tmp/stage6-7-seam-isolation.json \ + --raw-dir /tmp/stage6-7-seam-isolation-raw \ + --checkpoint +# 5. assemble the composite authority from the two raw sections +npx tsx tests/perf/assembleCompositeEvidence.ts \ + --general /tmp/time-to-answer-general.json \ + --stageFive /tmp/time-to-answer-stage5.json \ + --output /tmp/time-to-answer-baseline.json +# 6. recompute summaries from the raw samples (never trust supplied aggregates) +npm run perf:summarize -- --artifact /tmp/time-to-answer-baseline.json +# 7. compare a candidate against a reviewed baseline (fails closed) +npm run perf:time-to-answer -- \ + --metadata /tmp/time-to-answer-metadata.json \ + --output /tmp/time-to-answer-candidate.json \ + --baseline /tmp/time-to-answer-baseline.json +``` + +Prerequisites: a release daemon (`cargo build --release -p dockermap-daemon`), +Chromium for Playwright, and a built web app — the capture performs the contract, +production web, benchmark-mode application and module-probe builds itself, and +pins the artifacts it serves before measuring anything. Each benchmark entrypoint +owns every process it starts: the general capture owns the fixture Docker daemon, +the real daemon and API, the production and benchmark builds and real Chromium, +and the dedicated protocols own their own private fixture, daemon, API, server +and browser contexts. + +**Capture discipline.** The capture refuses to start from a dirty worktree, and +refuses to run if the metadata's `sourceRevision`, `harnessRevision` or +`methodologyVersion` does not match the checked-out commit and the contract. A +baseline is therefore always reproducible from a committed revision: the artifact +names both the product revision and the harness that measured it, and the design it +was measured under. Commit the harness **before** capturing — baseline 1 was +invalidated precisely because its harness existed only as uncommitted changes — and +run the focused smoke (`DOCKERMAP_BENCH_DEBUG=1` with `--fixtures`, which relaxes +only the run/sample counts for probing and can never emit an artifact) before +spending a full capture. + +### Warm-up calibration protocol (REJECTED — retained as audit evidence only) + +This collector is retained as audit evidence. Per-metric stationarity calibration is +**retired**: it does not gate Baseline-4, it supplies no warm-up count, and it is not a +step in the sequence above. The procedure below is recorded so that the retirement is +auditable, not because it is current. + +Calibration is an independent, bounded conditioning collector. It never calls the +frozen-count lookup, never enters baseline assembly or normal capture's frozen-count +preflight, and never emits or merges baseline raw evidence. For every +`warmed-repeated` end-to-end metric and each declared reference fixture +(`reference-25`, `reference-100`, `reference-250`), it retains exactly **60 ordered +finite observations** beginning at call zero. This includes daemon-attribution +metrics, browser/API-path metrics, and the module-probe metrics; the probe's ordinary +hidden two-call warm-up is disabled for calibration. + +For each fixture trace, candidates `w=2..45` compare the median of observations +`[w-2,w)` with the median of the following 15 observations `[w,w+15)`. A candidate is +stable only if the ratio is within the frozen **0.5–1.5x** band at that candidate and +every later eligible candidate. The fixture value is the earliest sustained `w`; the +metric value is the maximum fixture value plus the frozen safety margin **2**. The +result must have a complete following 15-observation window within the retained 60. +A failure is a calibration conflict: +the collector does not extrapolate, expand the window, retry toward a preferred point, +or select a fixture-specific baseline count. + +### Calibration capacity and superseded evidence + +Methodology-8 adopts fixed 60+15 conditioning after the final 60-observation +calibration disproved the premise that every metric supports one common sustained +stationarity gate. The calibration capacity, band, and margin are historical +diagnostic details; they are not a Baseline-4 prerequisite. + +The rejected 40-observation calibration remains retained evidence, not Baseline-4 +authority: `buildModelMs: stable w=25 -> frozen warm-up=27 -> requires 42 +observations`. That failure demonstrated insufficient protocol capacity; it does not +itself define the new window. + +The external calibration artifact stores its raw ordered cells, constants, pinned +environment, daemon-binary provenance, and complete per-metric derivation trace. +It is persisted with SHA-256 even on conflict, but is **REJECTED, +NON-AUTHORITATIVE** for Baseline-4: no result table may block or alter capture. + +No per-metric warm-up count is published. The calibration results were rejected and +never supplied a Baseline-4 parameter, so there is no result table to carry forward. + +Procedure notes: stages 8 and 10 run against the benchmark-only module probe +(`tests/perf/benchVite.config.mjs`, real production modules, real Chromium); +stages 6 and 7 run against the benchmark-mode application build +(`tests/perf/benchAppVite.config.mjs`); stages 11 and 12 run against the ordinary +production build. `tests/perf/browserProbe.js` is test-only instrumentation loaded +before product code. `.bench-dist` and `.bench-app-dist` are generated and +gitignored. Every capture also writes `.harness-evidence.json`, which is +not part of the closed artifact schema and carries: + +- the general capture's own observation records and its retained burn-in windows; + the dedicated Stage-6/7 seam-isolation control is NOT part of this file — it runs + as its own protocol with its own raw directory and its own companion record, and + its samples never become a Baseline-4 timing row; +- burn-in retention: for every ordinary warmed end-to-end cell the **complete** + `samples + 60` observation window in order, with fixed burn-in at indices 0–59 + and observations 60–74 proven equal to the recorded run; +- the retained `burnInObservations` map. + +The capture refuses to emit an artifact when an ordinary warmed window is missing, +shorter than `samples + 60`, retains anything other than the fixed first 60 +observations, or does not match the artifact. Browser and probe stages are held to +the same retained 60+15 structural rule. + +## Cold-start versus warmed-repeated stages + +The distinction is load-bearing, not descriptive, and the contract encodes it in +`TIME_TO_ANSWER_STAGE_KIND`: + +- **cold-start** — `daemonStartToListenerMs`, `listenerToFirstDockerModelMs`. The + first observation *is* the measurement, so nothing is discarded. +- **warmed-repeated** — every other stage. A repeated operation. The + daemon's first passes through the collection path are cold (its first refresh + runs before its listener binds). The protocol therefore declares a **FIXED + `TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS = 60` burn-in observations BEFORE + the capture and collects `samples + 60` observations, keeping the first 60 for + audit and the next 15 as the measured window + (`splitWarmedObservations` refuses a shorter window). With 15 recorded samples, + nearest-rank p95 *is* the maximum, so a surviving cold observation would + otherwise become the published number. +- **scenario cells** — a warmed stage measured on a scenario fixture + (`isScenarioCell`). They are declared in the closed matrix only for the fixtures + that construct the scenario. + +**Why sixty.** The final calibration showed no common sustained-stationarity rule +for all end-to-end metrics. Sixty is fixed before capture, is not chosen or +extended from observed values, and is equal conditioning rather than a claim that +all metrics reach steady state. + +**Stationarity is informational only.** Historical diagnostic calculations may be +retained for audit, but never invalidate a structurally valid run, move observation +61, or cause any sample to be dropped. + +Every burn-in observation is retained in the raw audit trail +(`burnInObservations`, keyed `fixture|stage|run`) with the whole observation +window, and none enters a recorded sample, summary, or promotion comparison. + +## Capture discipline + +The capture refuses to start from a dirty worktree and refuses to run when the +metadata's `sourceRevision` or `harnessRevision` does not match the checked-out +commits, so a baseline is always reproducible from a **committed** revision: + +- `sourceRevision` — the product revision the numbers describe. +- `harnessRevision` — the last commit touching `tests/perf` and the performance + contract, i.e. the harness that produced them. + +Reproducing a recorded baseline therefore requires checking out the revision the +artifact names; re-emitting metadata at a different commit produces a different +artifact by design. + +## Daemon binary provenance + +Stages 1-5 and 9 all come from the release daemon executable, so it is pinned: + +``` +cargo build --release --locked -p dockermap-daemon --manifest-path crates/Cargo.toml +sha256(crates/target/release/dockermap-daemon) # daemonBinarySha256 +``` + +`emit-metadata` performs that build and records `daemonBinarySha256`, +`daemonBinaryBuild` and `cargoRevision`. The capture verifies the digest **before** +the run and **again after** it — the daemon is spawned repeatedly during a long +capture, so a mid-run substitution or rebuild would otherwise be invisible — and +fails closed on mismatch or on a substituted executable +(`assertDaemonBinaryProvenance`). Both digests and the build command are recorded in +the harness evidence beside the artifact. This proves which binary *this* capture +executed; it is **provenance, not a promotion compatibility requirement** (see +"Promotion rules"). + +## Stage 5 — the deterministic poll-phase sweep + +The baseline measures today's real publication→observation mechanism **including +its poll wait**, and it *drives* the publication phase instead of hoping for one. + +Baseline 3 disproved the earlier hope: it slept a uniform random delay before each +trigger, and reference-25's 45 samples still occupied a **120 ms band of the +2000 ms interval (6.0 %)** with consecutive differences under 19 ms — because both +the daemon's refresh loop and the API's poller run on fixed 2 s cycles, so the +measured gap was their phase offset rather than a sample of any distribution. + +The declared design (`timeToAnswerPollPhase.ts`, methodology revision 2): + +- the poll interval is divided into **10 equal divisions**; phase `p` places the + publication at `(p + 0.5) × interval / 10`, so the intended latency is + `interval − that offset` and no phase sits on a poll tick boundary (where a + publication is inherently ambiguous); +- each controlled 15-sample run sweeps the phases **ascending** and repeats phases + 0–4. Every raw sample's phase is recoverable from its position in its run; every + phase is represented at least twice, and repeated phases contribute all their + observations to that phase's median without giving that phase extra weight in the + normalized result; +- the benchmark-only fixture proxy **arms a unique trigger identity before the +measurement window**, witnesses and retains its exact daemon revision, then releases +that revision at the declared offset after an actual API `/daemon/health` poll. The +following real 2000 ms API poll observes it; no predicted daemon grid is used; +- each sample then **verifies** itself: the observed publication must match the + prediction, the observed latency must land on the declared phase within a **90 ms + tolerance** (the grid step is 200 ms, so adjacent phases stay distinguishable), and + the observation must have arrived through the real API stream. The capture also + asserts that the application page never contacted the daemon directly. + +Before accepting a capture, `assertPollPhaseSweep` requires that every declared +phase is represented at least twice, that no sample's phase was uncontrolled, that +all observations travelled the real poller path, that the sweep spans at least half +the interval, and that the earliest declared phase is faster than the latest by at +least half an interval. **A sweep confined to a narrow band is rejected** — that is +a RED test, not an aspiration. + +The reported figures are: + +- the **phase curve** — declared phase → observed latency (min, max, median per + phase), the most direct statement about the existing mechanism; and +- a **phase-normalized p95**, computed from the predeclared uniform grid by taking + observed median at each declared phase and then nearest-rank p95 over those + medians. It is the controlled stage-5 review and promotion authority; ordinary + per-run p95 remains diagnostic only. It weights the declared phases uniformly to characterise the latency the + fixed polling mechanism imposes. **It does not claim that real host publications + occur uniformly across poll phase**, and it is neither an observed user-traffic + distribution nor network latency. + +`publicationToNodeObservationMs` must never be described as network latency, and +the production cadence is deliberately unchanged: removing this floor is #337's +work, not this issue's. + +### Controlled Stage-5 cells + +The composite sources all six `publicationToNodeObservationMs` rows from +`time-to-answer-stage5.json` with the `controlled-poll-phase` protocol: +`reference-25`, `reference-100`, `reference-250`, +`provider-only-revision-change`, `docker-topology-change`, and +`unavailable-optional-provider`. Every row has a declared phase and its reviewed +aggregation is the phase-normalized p95 derived by +`derivedTimeToAnswerSummaries`. + +`POLL_PHASE_CONTROLLED_FIXTURES` and `isPhaseControlledFixture` select the +reference fixtures plus `docker-topology-change` for the summary's dedicated +per-phase curve table. That table selection does not exempt the two provider-driven +fixtures from their controlled-poll-phase composite rows or phase-normalized +reviewed value. + +The provider-driven fixtures derive their new revision from the daemon's fixed +host-provider scheduler while the API SSE poller remains pinned at 2000 ms. This is +structural context, not a phase-ownership exemption; their values are not real-user +latency, random production latency, or network latency. + +## Baseline-4 composite capture + +Baseline 4 is a composite artifact with three raw evidence sections. The general +capture records only `end-to-end` cells, including the normal Stage-6 and +Stage-7 timing rows. The dedicated Stage-5 sub-benchmark +records every `publicationToNodeObservationMs` cell with +`controlled-poll-phase`; it is the sole owner of the arm → mark → trigger → +identity-ack protocol. The sections are not pooled: every composite record +names its fixture, stage, measurement protocol, source evidence file and +committed checkpoint SHA. The dedicated Stage-6/7 seam-isolation protocol is +`controlled-stage6-stage7-seam-isolation`: it observes acceptance at the real +`useSystemModel` coherent snapshot/runtime-map seam, requires an internally +coherent accepted pair, and applies its 250 ms delay only after that acceptance. +Its supporting-evidence companion is **FAIL** at this checkpoint: `[independence] +FAIL: Error: control delay did not begin after the acceptance timestamp`. It records +the exact limitation `publication-level causal identity unavailable` and +`validatesDaemonToBrowserAttribution: false`; it makes no daemon-publication +attribution claim. Its samples are never Baseline-4 timing +observations and cannot replace or contaminate the normal end-to-end Stage-6/7 +rows. Assembly rejects a missing Stage-5 section, duplicate declared cell, wrong +protocol, mismatched methodology/checkpoint, or invalid supplied seam-isolation +evidence; absence of this supporting control does not prevent the normal 60 +burn-in + 15 measured Baseline-4 timings from existing. The seam-isolation entrypoint is self-orchestrating under the trusted capture +invocation: it owns a private fixture, real daemon/API/SSE path, benchmark build, +fresh browser contexts, and `finally` teardown. It retains diagnostics only under +the dedicated protocol directory and refuses partial output. + +The Stage-5 metric and phase-normalized authority are unchanged. Conditioning is a +fixed 60-observation burn-in followed by 15 measured observations, with every +burn-in retained and no stationarity gate. The phase grid, 90 ms tolerance, +Stage-6/7 semantics and the production 2000 ms poller are also unchanged. + +Baseline 1, baseline 2 **and baseline 3** are **REJECTED historical attempts** and +are not the authority for anything. Their artifacts are kept outside the repository +in `/srv/jonas/evidence/dockermap/` (baselines 1 and 2) and +`/srv/jonas/evidence/dockermap/time-to-answer/` (baseline 3, with its harness +evidence). Their numbers may be cited only to explain methodology changes, never as +current measurements and never for promotion gating. Baseline 3 was rejected +because its stage-5 samples were a phase-locked sawtooth rather than a sweep of the +poll interval, its single discarded warm-up still left a 2.15×-of-median +observation inside the measured window, it gated promotion on the rebuilt daemon +digest, and it documented a post-run binary verification the code did not perform. + +## Current state of this slice + +Complete and enforced by tests: + +- the closed contract, the 12 stages and their buckets, the fixture set, the + 44-cell fixture-by-stage matrix, the environment allowlist (including the + effective SSE poll interval), raw-sample validation, the summary math and the + promotion gate; +- the deterministic fixture topology (whose generation delta is product-visible, + so the stage-7 expected-content check is discriminating) and the fixture Docker + daemon, proven against the real daemon build; +- the inert bench-only stage attribution hook for `dockerObservationMs`, + `composeEnrichmentMs` and `findingsDerivationMs`; +- the stage-6 coherent-model acceptance seam in real product source, compiled out + of the production build and compiled into the benchmark-mode application build, + with the stage-7 expected-content + single-frame end condition and the + chain-of-custody check from daemon revision to rendered content; +- the dedicated controlled Stage-5 protocol, the sole owner of arm → mark → + trigger → identity-ack; it has no Stage-5 timing cell in the general capture; +- the dedicated, self-orchestrating Stage-6/7 seam-isolation control as supporting + evidence, unit-tested against its RED cases; +- the fixed 60-observation burn-in plus 15 measured conditioning, with every + burn-in retained and no stationarity gate; +- the capture's runtime premise assertions, and the documented general capture, + Stage-5 capture, assembly, and summary commands with their browser probes and + environment emitter; +- the promotion RED-checks (`timeToAnswerPromotion.test.ts`), the independence + RED-checks (`timeToAnswerIndependence.test.ts`) and the production isolation + proof (`productionIsolation.test.mjs`); +- the phase-sweep RED-checks (`timeToAnswerPollPhase.test.ts`) alongside + `npm run perf:phase-control`, which records intended and observed phase, trigger + identity, phase error and grid span and fails RED on a substituted or uncontrolled + publication; and +- `npm run test:perf` wired into `npm run check:js`. + +`docs/testing/TIME_TO_ANSWER_BASELINE.md` is the baseline record. Baseline 3 (from +committed revision `cf77e8ba`) was **REJECTED** in round-3 review and is not the +authority for anything: its stage-5 sweep did not sweep, its warm-up policy left a +cold observation inside the measured window, its promotion gate treated the rebuilt +daemon digest as a compatibility key, and it documented a post-run binary +verification the code did not perform. diff --git a/package.json b/package.json index 17a73ea2..126e17b1 100644 --- a/package.json +++ b/package.json @@ -13,9 +13,9 @@ "dev:stack": "./scripts/run-dev-stack.sh", "build": "npm run build --workspaces --if-present", "build:deploy": "VITE_API_BASE_URL=\"\" npm run build && cargo build --release -p dockermap-daemon -p dockermap-docker-gateway --manifest-path crates/Cargo.toml", - "typecheck": "npm run typecheck --workspaces --if-present", +"typecheck": "npm run typecheck --workspaces --if-present && tsc --project tests/perf/tsconfig.json", "audit": "npm audit --omit=dev", - "check:js": "npm run check:version && npm run audit && npm run typecheck && npm run build && npm run check:contracts && npm run test:version && npm run test:deployment && npm run test:js", + "check:js": "npm run check:version && npm run audit && npm run typecheck && npm run build && npm run check:contracts && npm run test:version && npm run test:perf && npm run test:deployment && npm run test:js", "ci:js": "npm ci && npm run check:js", "fmt:rust": "cargo fmt --manifest-path crates/Cargo.toml --all", "fmt:rust:check": "cargo fmt --manifest-path crates/Cargo.toml --all -- --check", @@ -30,6 +30,15 @@ "test:web": "npm run test --workspace @dockermap/web --if-present", "test:contracts": "npm run test --workspace @dockermap/contracts --if-present", "test:version": "node --test scripts/check-version-authority.test.mjs scripts/package-release.test.mjs", + "test:perf": "node --test tests/perf/*.test.mjs", + "perf:time-to-answer": "tsx tests/perf/capture.ts", + "perf:calibrate-time-to-answer": "tsx tests/perf/capture.ts", + "perf:preconditioning": "tsx tests/perf/preconditioning.ts", +"perf:phase-control": "tsx tests/perf/phaseControl.ts", + "perf:stage-five": "tsx tests/perf/captureStageFive.ts", + "perf:independence": "tsx tests/perf/captureIndependence.ts", + "perf:summarize": "tsx tests/perf/summarize.ts", + "perf:metadata": "node tests/perf/emit-metadata.mjs", "test:deployment": "node --test scripts/check-systemd-profile.test.mjs scripts/check-supply-chain-baseline.test.mjs", "generate:contracts": "cargo run -p dockermap-core --bin generate-contract-schemas --manifest-path crates/Cargo.toml -- && node scripts/generate-rust-contract-types.mjs", "check:contracts": "node scripts/check-rust-contract-schemas.mjs && node scripts/generate-rust-contract-types.mjs --check && node --test scripts/generate-rust-contract-types.test.mjs && npm run test:contracts", diff --git a/tests/perf/assembleCompositeEvidence.ts b/tests/perf/assembleCompositeEvidence.ts new file mode 100644 index 00000000..4da71471 --- /dev/null +++ b/tests/perf/assembleCompositeEvidence.ts @@ -0,0 +1,36 @@ +/** Composite Baseline-4 assembly. Sections remain separate until validation. */ +import { readFileSync, writeFileSync } from "node:fs"; +import { TIME_TO_ANSWER_BASELINE, TIME_TO_ANSWER_CONTROLLED_POLL_MATRIX, TIME_TO_ANSWER_END_TO_END_MATRIX, TIME_TO_ANSWER_METHODOLOGY, validateTimeToAnswerEvidence } from "../../apps/web/src/lib/performance/timeToAnswerEvidence"; + +export function assembleCompositeEvidence(general: any, stageFive: any, stageSixSeven: any | null, sources: { general: string; stageFive: string; stageSixSeven?: string }) { + if (general.environment?.methodologyVersion !== TIME_TO_ANSWER_METHODOLOGY || stageFive.methodologyVersion !== TIME_TO_ANSWER_METHODOLOGY || (stageSixSeven && stageSixSeven.methodologyVersion !== TIME_TO_ANSWER_METHODOLOGY)) throw new Error("raw sections do not share the Baseline-4 methodology"); + if (!stageFive.checkpointSha || (stageSixSeven && stageFive.checkpointSha !== stageSixSeven.checkpointSha) || general.environment?.sourceRevision !== stageFive.checkpointSha) throw new Error("raw sections do not share a committed checkpoint"); + const stageFiveFixtures = new Set(TIME_TO_ANSWER_CONTROLLED_POLL_MATRIX.map((cell) => cell.fixture)); + const stageFiveRecords = Object.entries(stageFive.cells ?? {}).map(([fixture, samples]: [string, any]) => ({ + fixture, stage: "publicationToNodeObservationMs", measurementProtocol: "controlled-poll-phase", sourceEvidenceFile: sources.stageFive, checkpointSha: stageFive.checkpointSha, + runs: Array.from({ length: 3 }, (_, run) => samples.filter((sample: any) => sample.run === run).sort((a: any, b: any) => a.sample - b.sample).map((sample: any) => sample.observedLatencyMs)) +})); + if (stageFiveRecords.length !== stageFiveFixtures.size || stageFiveRecords.some((record) => !stageFiveFixtures.delete(record.fixture))) throw new Error("Stage-5 controlled evidence does not cover the required fixtures exactly once"); + const controlledFixtures = new Set(TIME_TO_ANSWER_END_TO_END_MATRIX.filter((cell) => cell.stage === "notificationToCoherentModelMs").map((cell) => cell.fixture).filter((fixture) => TIME_TO_ANSWER_END_TO_END_MATRIX.some((cell) => cell.fixture === fixture && cell.stage === "coherentModelToUsefulRenderMs"))); + for (const record of general.records ?? []) { + if (!TIME_TO_ANSWER_END_TO_END_MATRIX.some((cell) => cell.fixture === record.fixture && cell.stage === record.stage)) throw new Error(`general evidence contains a non-end-to-end cell: ${record.fixture}/${record.stage}`); + } + const generalRecords = (general.records ?? []).map((record: any) => ({ ...record, measurementProtocol: "end-to-end", sourceEvidenceFile: sources.general, checkpointSha: stageFive.checkpointSha })); + if (stageSixSeven === null) return validateTimeToAnswerEvidence({ baseline: TIME_TO_ANSWER_BASELINE, environment: general.environment, records: [...generalRecords, ...stageFiveRecords] }); + if (stageSixSeven.measurementProtocol !== "controlled-stage6-stage7-seam-isolation" || !Array.isArray(stageSixSeven.cells) || !stageSixSeven.independenceEvidence || stageSixSeven.independenceEvidence.verdict !== "PASS" || stageSixSeven.independenceEvidence.limitation !== "publication-level causal identity unavailable" || stageSixSeven.independenceEvidence.validatesDaemonToBrowserAttribution !== false) throw new Error("Stage-6/7 seam-isolation evidence is invalid"); +const expectedFixtures = [...controlledFixtures].sort(); +const receivedFixtures = stageSixSeven.cells.map((cell: any) => cell.fixture).sort(); +if (JSON.stringify(receivedFixtures) !== JSON.stringify(expectedFixtures)) throw new Error("Stage-6/7 independence evidence does not cover the required fixtures exactly once"); +for (const cell of stageSixSeven.cells) { + if (!Array.isArray(cell.normalStageSixRuns) || !Array.isArray(cell.normalStageSevenRuns) || !cell.controlEvidence || cell.controlEvidence.delayMs !== 250 || cell.controlEvidence.acceptanceSeam !== "observed" || !Array.isArray(cell.controlEvidence.samples) || "triggerRevision" in cell.controlEvidence || cell.controlEvidence.samples.some((sample: any) => !sample.acceptedRevision || sample.snapshotRevision !== sample.acceptedRevision || sample.runtimeMapRevision !== sample.acceptedRevision || sample.delayAppliedAfterAcceptance !== true || sample.delayStartedAfterAcceptance !== true || "triggerRevision" in sample || "triggerId" in sample || "acknowledgement" in sample)) throw new Error(`Stage-6/7 seam-isolation evidence is incomplete or makes a publication-attribution claim for ${cell.fixture}`); +} + return validateTimeToAnswerEvidence({ baseline: TIME_TO_ANSWER_BASELINE, environment: general.environment, records: [...generalRecords, ...stageFiveRecords] }); +} + +const args = Object.fromEntries(process.argv.slice(2).flatMap((value, index, all) => value.startsWith("--") ? [[value.slice(2), all[index + 1] ?? ""]] : [])); +if (!args.general || !args.stageFive || !args.output) throw new Error("--general, --stageFive and --output are required"); +const general = JSON.parse(readFileSync(args.general, "utf8")); +const stageFive = JSON.parse(readFileSync(args.stageFive, "utf8")); +const stageSixSeven = args.stageSixSeven ? JSON.parse(readFileSync(args.stageSixSeven, "utf8")) : null; +writeFileSync(args.output, `${JSON.stringify(assembleCompositeEvidence(general, stageFive, stageSixSeven, { general: args.general, stageFive: args.stageFive, stageSixSeven: args.stageSixSeven }), null, 2)}\n`); +writeFileSync(`${args.output}.supporting-evidence.json`, `${JSON.stringify({ stageSixSevenSeamIsolation: stageSixSeven ? { verdict: stageSixSeven.independenceEvidence?.verdict ?? "FAIL", limitation: "publication-level causal identity unavailable", validatesDaemonToBrowserAttribution: false, sourceEvidenceFile: args.stageSixSeven } : { verdict: "FAIL", limitation: "publication-level causal identity unavailable", validatesDaemonToBrowserAttribution: false } }, null, 2)}\n`); diff --git a/tests/perf/benchAppVite.config.mjs b/tests/perf/benchAppVite.config.mjs new file mode 100644 index 00000000..b1f7857d --- /dev/null +++ b/tests/perf/benchAppVite.config.mjs @@ -0,0 +1,37 @@ +import { fileURLToPath } from "node:url"; +import { defineConfig } from "vite"; +import react from "@vitejs/plugin-react"; + +/** + * BENCHMARK-MODE APPLICATION build (#335). + * + * It builds the REAL production application (`apps/web`) with the + * benchmark-only acceptance seam compiled IN: + * + * __DOCKERMAP_BENCH_ACCEPTANCE__ = "true" + * + * That is the only difference from the shipped build. The application source is + * not copied, forked or reimplemented — the seam is the same code the product + * build compiles out (`apps/web/src/lib/performance/modelAcceptance.tsx`), and + * the capture uses this build only for the browser stages that need to observe + * coherent-model acceptance (stages 6 and 7) and their independence control. + * Every other browser stage (production bundle/startup, Cmd-K) is measured on + * the ordinary production build (`apps/web/dist`) served alongside it. + * + * It is read exclusively by `npm run perf:time-to-answer`; no production build + * path references this file. + */ +const appRoot = fileURLToPath(new URL("../../apps/web", import.meta.url)); + +export default defineConfig({ + root: appRoot, + base: "./", + plugins: [react()], + define: { __DOCKERMAP_BENCH_ACCEPTANCE__: "true" }, + build: { + outDir: fileURLToPath(new URL("./.bench-app-dist", import.meta.url)), + emptyOutDir: true, + target: "es2022", + sourcemap: false + } +}); diff --git a/tests/perf/benchVite.config.mjs b/tests/perf/benchVite.config.mjs new file mode 100644 index 00000000..7bbe8dd0 --- /dev/null +++ b/tests/perf/benchVite.config.mjs @@ -0,0 +1,29 @@ +import { defineConfig } from "vite"; +import { fileURLToPath } from "node:url"; + +/** + * BENCHMARK-ONLY Vite config (#335). + * + * It builds `tests/perf/probe/` — an entry that imports the real production + * modules — into `tests/perf/.bench-dist`. It is used exclusively by + * `npm run perf:time-to-answer`. + * + * The ordinary production build (`npm run build --workspace @dockermap/web`) + * does not read this file and does not include this entry. A hostile regression + * test asserts that the production artifact contains no probe identifiers. + */ +const probeRoot = fileURLToPath(new URL("./probe", import.meta.url)); + +export default defineConfig({ + root: probeRoot, + base: "./", + build: { + outDir: fileURLToPath(new URL("./.bench-dist", import.meta.url)), + emptyOutDir: true, + target: "es2022", + sourcemap: false, + rollupOptions: { + input: fileURLToPath(new URL("./probe/index.html", import.meta.url)) + } + } +}); diff --git a/tests/perf/browserLifecycle.mjs b/tests/perf/browserLifecycle.mjs new file mode 100644 index 00000000..b5102deb --- /dev/null +++ b/tests/perf/browserLifecycle.mjs @@ -0,0 +1,31 @@ +/** + * Own the browser lifetime for one controlled capture run. + * + * A capture run may exercise several fixtures, but it must not inherit browser + * pages, connections, caches, or renderer state from an earlier run. + */ +export async function withFreshBrowserRuns({ runs, launch, run, lifecycle = (..._args) => {} }) { + const created = new Set(); + const closed = new Set(); + for (let runIndex = 0; runIndex < runs; runIndex += 1) { + const browser = await launch(); + const id = `browser-${runIndex}`; + created.add(id); + lifecycle("create", "browser", id); + try { + await run(browser, runIndex); + } finally { + try { + await browser.close(); + closed.add(id); + lifecycle("closure", "browser", id); +} catch (error) { +lifecycle("exception", "browser_close", error); +throw error; + } + } + } + if (created.size !== closed.size || [...created].some((id) => !closed.has(id))) { + throw new Error(`browser lifecycle leak: created ${created.size}, closed ${closed.size}`); + } +} diff --git a/tests/perf/browserLifecycle.test.mjs b/tests/perf/browserLifecycle.test.mjs new file mode 100644 index 00000000..9efb7393 --- /dev/null +++ b/tests/perf/browserLifecycle.test.mjs @@ -0,0 +1,93 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import { withFreshBrowserRuns } from "./browserLifecycle.mjs"; + +test("repeated control-shaped runs start with a fresh browser and close it after every run", async () => { + const browsers = []; + const rendered = []; + await withFreshBrowserRuns({ + runs: 21, + launch: async () => { + const browser = { pageValue: "0", closed: false, async close() { this.closed = true; } }; + browsers.push(browser); + return browser; + }, + run: async (browser, runIndex) => { + // This is the stale-Home failure shape: a reused page would retain the + // previous control's value instead of beginning from the fixture baseline. + assert.equal(browser.pageValue, "0"); + browser.pageValue = String(runIndex + 1); + rendered.push(browser.pageValue); + } + }); + assert.deepEqual(rendered, Array.from({ length: 21 }, (_, index) => String(index + 1))); + assert.equal(new Set(browsers).size, 21); + assert.ok(browsers.every((browser) => browser.closed)); +}); + +test("a failed run still closes its browser before the next clean run", async () => { + const browsers = []; + await assert.rejects( + withFreshBrowserRuns({ + runs: 2, + launch: async () => { + const browser = { closed: false, async close() { this.closed = true; } }; + browsers.push(browser); + return browser; + }, + run: async () => { throw new Error("control failed"); } + }), + /control failed/ + ); + assert.equal(browsers.length, 1); +assert.equal(browsers[0].closed, true); +}); + +test("reports browser closure without changing fresh-browser ownership", async () => { + const events = []; + await withFreshBrowserRuns({ + runs: 1, + launch: async () => ({ async close() {} }), + run: async () => {}, + lifecycle: (...event) => events.push(event) + }); + assert.deepEqual(events, [["create", "browser", "browser-0"], ["closure", "browser", "browser-0"]]); +}); + +test("a sustained preconditioning run balances browser and stage-5 reader lifecycles", async () => { + const { runSequentialPreconditioning } = await import("./preconditioningLifecycle.mjs"); + const result = await runSequentialPreconditioning({ runs: 12, launch: async () => ({ async close() {} }) }); + assert.equal(result.events.filter((event) => event.event === "create").length, + result.events.filter((event) => event.event === "teardown").length); + assert.ok(result.events.some((event) => event.generation === 16)); + assert.ok(result.events.some((event) => event.generation === 17)); + assert.ok(result.events.some((event) => event.generation === 18)); +}); + +test("a leaked stage-5 reader fails the sustained preconditioning lifecycle gate", async () => { + const { runSequentialPreconditioning } = await import("./preconditioningLifecycle.mjs"); + await assert.rejects( + runSequentialPreconditioning({ + runs: 9, + launch: async () => ({ async close() {} }), + createReader: async () => ({ async cancel() { throw new Error("reader leak"); } }) + }), + /reader leak/ + ); +}); + +test("preconditioning rejects a depth outside its sustained-run contract", async () => { + const { runSequentialPreconditioning } = await import("./preconditioningLifecycle.mjs"); + await assert.rejects(runSequentialPreconditioning({ runs: 8 }), /requires 9-16 sequential runs/); +}); + +test("fails immediately when a browser close leaks", async () => { + await assert.rejects( + withFreshBrowserRuns({ + runs: 1, + launch: async () => ({ async close() { throw new Error("close failed"); } }), + run: async () => {} + }), + /close failed/ + ); +}); diff --git a/tests/perf/browserProbe.js b/tests/perf/browserProbe.js new file mode 100644 index 00000000..cdae4137 --- /dev/null +++ b/tests/perf/browserProbe.js @@ -0,0 +1,626 @@ +/* + * Test-only browser instrumentation for the time-to-answer benchmark (#335). + * + * Loaded by the capture harness with `page.addInitScript({ path })`, so it runs + * before any product code. It records timestamps only: + * + * - when the real stream notifies the browser of a new model revision; + * - when the application ACCEPTS a coherent model (read from the + * benchmark-only acceptance sink the real application seam writes to — this + * is application state, never a DOM mutation); + * - when the product commits model-derived DOM text, together with the + * accepted revision the document was stamped with at that commit. + * + * It is plain JavaScript on purpose. A TS-authored init script is serialised + * through esbuild's `keepNames` helper (__name), which does not exist in the + * page realm and aborts the script — see #335. + * + * Nothing here is part of the production bundle and nothing is uploaded. + */ +(() => { + const bench = { + notifyAt: 0, + notifyRevision: "", + notifyLog: [], + fetchLog: [], + requestOrigins: [], + streamUrl: "", + opens: 0, + errors: 0, + events: 0, + lastData: "", + installed: false, + initError: "", + observerError: "", + commits: [], + arm: null + }; + window.__dockermapBench = bench; + + /* + * Fetch attribution. The application does NOT accept whatever the stream + * announces: it accepts the coherent snapshot/runtime-map pair its own fetches + * returned. The daemon is read per request, so `/daemon/health` (what the stream + * carries) and `/daemon/snapshot` (what the pair carries) can legitimately hold + * different revisions while the host is churning. The harness therefore records + * which revision each paired fetch actually delivered, so stage 6 can start at + * the notification that caused THAT fetch cycle instead of assuming the accepted + * revision was announced by the stream. + */ + try { + const originalFetch = window.fetch; + if (typeof originalFetch === "function") { + window.fetch = function (...callArgs) { + const input = callArgs[0]; + const url = typeof input === "string" ? input : String((input && input.url) || ""); + const paired = /\/api\/(snapshot|runtime\/map)(?:[?#]|$)/.test(url); + const startedAt = performance.now(); + // Every request origin the app used, so the harness can prove the page never + // reached the daemon directly: stage 5/6 must travel the API's real poller. + try { + const resolved = new URL(url, location.href).origin; + if (resolved && !bench.requestOrigins.includes(resolved)) { + bench.requestOrigins.push(resolved); + if (bench.requestOrigins.length > 64) bench.requestOrigins.shift(); + } + } catch (error) { + // non-URL request target: ignore + } + const result = originalFetch.apply(this, callArgs); + if (paired && result && typeof result.then === "function") { + result + .then((response) => { + try { + const clone = response.clone(); + return clone.json().then((payload) => { + bench.fetchLog.push({ + url: /runtime\/map/.test(url) ? "runtime-map" : "snapshot", + startedAt, + at: performance.now(), + revision: payload && payload.modelRevision ? String(payload.modelRevision) : "" + }); + if (bench.fetchLog.length > 512) bench.fetchLog.splice(0, 256); + }); + } catch (error) { + return undefined; + } + }) + .catch(() => undefined); + } + return result; + }; + } + } catch (error) { + bench.initError = bench.initError || String(error); + } + + try { + const Original = window.EventSource; + if (typeof Original !== "function") { + bench.initError = "EventSource is not constructible in this realm"; + } else { + // The product's stream is the real EventSource; this subclass only + // timestamps what the product already receives. + const BenchEventSource = function (url, init) { + const source = new Original(url, init); + bench.streamUrl = String(url); + const record = (event) => { + bench.events += 1; + if (!bench.lastData) bench.lastData = String((event && event.data) || "").slice(0, 160); + let revision = ""; + try { + revision = (JSON.parse((event && event.data) || "{}").modelRevision) || ""; + } catch (error) { + revision = ""; + } + // Timestamp every NEW notified revision: a publication can advance more + // than once per sample (provider state and inventory can both move), so + // the notification a later acceptance belongs to must be recoverable + // rather than assumed to be the latest one. + if (revision && revision !== bench.notifyRevision) { + bench.notifyAt = performance.now(); + bench.notifyRevision = revision; + bench.notifyLog.push({ at: bench.notifyAt, revision }); + if (bench.notifyLog.length > 256) bench.notifyLog.splice(0, 128); + } + }; + source.addEventListener("open", () => { + bench.opens += 1; + }); + source.addEventListener("error", () => { + bench.errors += 1; + }); + // The product names its event "snapshot"; "message" is kept so the probe + // still observes the notification if that ever changes. + source.addEventListener("snapshot", record); + source.addEventListener("message", record); + return source; + }; + BenchEventSource.prototype = Original.prototype; + Object.defineProperty(BenchEventSource, "name", { value: "EventSource" }); + Object.defineProperty(window, "EventSource", { + configurable: true, + writable: true, + value: BenchEventSource + }); + bench.installed = window.EventSource === BenchEventSource; + } + } catch (error) { + bench.initError = String(error); + } + + const readMetric = (label) => { + const metrics = Array.from(document.querySelectorAll("main .story .metric")); + for (const metric of metrics) { + const name = metric.querySelector(".metric-label"); + if (name && (name.textContent || "").trim() === label) { + const value = metric.querySelector(".metric-value"); + return value ? (value.textContent || "").trim() : ""; + } + } + return null; + }; + + const acceptedRevision = () => { + const root = document.documentElement; + return (root && root.dataset && root.dataset.dockermapAcceptedRevision) || ""; + }; + + try { + const observer = new MutationObserver((records) => { + for (const record of records) { + const node = record.target; + const element = node instanceof Element ? node : node.parentElement; + // The Home content regions. `main .story` is the metrics band; `main + // .stack` is Home's right column (attention list, map preview, feed). + const inStory = Boolean(element && element.closest("main .story")); + const inHome = Boolean(element && element.closest("main .story, main .stack")); + // A commit only counts as model content reaching the DOM if it alters + // rendered text. Attribute-only or node-shuffling churn does not. + let textChanged = record.type === "characterData"; + if (!textChanged) { + const list = (record.addedNodes || []).length + ? record.addedNodes + : record.removedNodes || []; + for (const added of list) { + const text = added.textContent || ""; + if (text.trim() !== "") { + textChanged = true; + break; + } + } + } + const commit = { at: performance.now(), inHome, inStory, textChanged, revision: "" }; + if (textChanged && inHome) { + // Stamped by the application in the same commit that rendered the + // accepted model, and the live metric band value at that commit. + commit.revision = acceptedRevision(); + commit.storyValue = readMetric("Offline"); + commit.servicesValue = readMetric("Services"); + } + bench.commits.push(commit); + if (bench.commits.length > 5000) bench.commits.splice(0, 2500); + } + }); + const start = () => { + try { + observer.observe(document.documentElement || document, { + childList: true, + subtree: true, + characterData: true + }); + } catch (error) { + bench.observerError = String(error); + } + }; + if (document.documentElement) start(); + else document.addEventListener("readystatechange", start, { once: true }); + } catch (error) { + bench.observerError = String(error); + } + + const frame = () => new Promise((done) => requestAnimationFrame(done)); + const acceptanceSink = () => window.__dockermapBenchAcceptanceSink || []; + const diagnostic = () => + "opens=" + + bench.opens + + " errors=" + + bench.errors + + " events=" + + bench.events + + " installed=" + + bench.installed + + " esType=" + + typeof window.EventSource + + " initError=" + + bench.initError + + " observerError=" + + bench.observerError + + " accepted=" + + acceptanceSink().length; + + /* + * Attribute an accepted revision to the notification that caused it: the paired + * fetch that DELIVERED that revision, then the browser notification that + * preceded that fetch's start. Fails closed with both logs when the chain cannot + * be established. + */ + const attributeNotification = (revision, acceptedAt, notBeforeAt = 0) => { + const delivered = bench.fetchLog + .filter((entry) => entry.revision === revision && entry.at <= acceptedAt && entry.startedAt >= notBeforeAt) + .pop(); + if (!delivered) { + return { + error: + "no paired API fetch delivered the accepted revision " + + revision + + " before acceptance after the trigger checkpoint " + + notBeforeAt + + " (fetch=" + + JSON.stringify(bench.fetchLog.slice(-6)) + + ")" + }; + } + const notified = bench.notifyLog + .filter((entry) => entry.at <= delivered.startedAt && entry.at >= notBeforeAt) + .pop(); + if (!notified) { + return { + error: + "no browser notification preceded the fetch cycle that delivered the accepted revision " + + revision + + " after the trigger checkpoint " + + notBeforeAt + + " (notify=" + + JSON.stringify(bench.notifyLog.slice(-6)) + + ")" + }; + } + return { + notifyAt: notified.at, + notifiedRevision: notified.revision, + fetchStartedAt: delivered.startedAt, + fetchDeliveredAt: delivered.at, + deliveredBy: delivered.url + }; + }; + + /* + * Measurement helpers. The harness calls these by NAME through a raw string + * expression (`page.evaluate("window.__dockermapBenchHelpers...")`), because a + * TS-authored function is re-emitted with esbuild's `__name` helper that does + * not exist in the page realm. + */ + window.__dockermapBenchHelpers = { + /* + * Arm the stage-6/7 measurement BEFORE the harness triggers a publication + * change, then await it afterwards. Arming records the pre-change Home + * metric value, so "the DOM changed" is measured rather than assumed. + */ + armModelAcceptance(input) { + const mode = input.mode === "acceptance-only" ? "acceptance-only" : "content"; +const arm = { +mode, +seamIsolation: Boolean(input.seamIsolation), + previousSeq: Number(input.previousSeq) || 0, + limit: Number(input.limit) || 60000, + metricLabel: String(input.metricLabel || "Offline"), + expectedMetricValue: String(input.expectedMetricValue || ""), + awaitPublicationTrigger: Boolean(input.awaitPublicationTrigger), + beforeMetricValue: readMetric(String(input.metricLabel || "Offline")), + startedAt: performance.now(), + armed: true, + result: null, + error: null + }; + arm.trigger = null; + arm.task = (async () => { + let deadline = arm.startedAt + arm.limit; + if (arm.awaitPublicationTrigger) { + while (!arm.trigger && performance.now() < deadline) await frame(); + if (!arm.trigger) { + throw new Error( + "the model acceptance probe was armed but the publication trigger was never marked " + + "(arm=" + JSON.stringify({ startedAt: arm.startedAt, previousSeq: arm.previousSeq }) + "; " + diagnostic() + ")" + ); + } + // The measurement deadline belongs to the publication being measured, not + // to the short pre-trigger arming handshake. + deadline = performance.now() + arm.limit; + } + const trigger = arm.trigger || { at: 0, acceptedSequence: arm.previousSeq, revision: "", fetchLogLength: 0, notifyLogLength: 0 }; + /* + * acceptance-only: fixtures whose published revision carries NO inventory + * change (provider state alone moved). Stage 6 is "notification -> coherent + * model accepted", which needs no DOM content, and stage 7 is not declared + * for them — requiring a Home repaint there would be an empty number. + */ + if (arm.mode === "acceptance-only") { + let accepted = null; + while (performance.now() < deadline && !accepted) { + accepted = acceptanceSink().find((entry) => entry.revision && entry.snapshotRevision === entry.revision && entry.runtimeMapRevision === entry.revision && entry.seq > trigger.acceptedSequence) || null; + if (!accepted) await frame(); + } + if (!accepted) throw new Error("no accepted coherent model was observed (" + diagnostic() + ")"); + const attribution = attributeNotification(accepted.revision, accepted.at, trigger.at); + if (attribution.error) throw new Error(attribution.error + " (" + diagnostic() + ")"); + return { + notificationToCoherentModelMs: accepted.at - attribution.notifyAt, + coherentModelToUsefulRenderMs: null, + acceptedRevision: accepted.revision, + snapshotRevision: accepted.snapshotRevision, + runtimeMapRevision: accepted.runtimeMapRevision, + acceptedSequence: accepted.seq, + notifiedRevision: attribution.notifiedRevision, + latestNotifiedRevision: bench.notifyRevision, + fetchDeliveredBy: attribution.deliveredBy, + fetchStartedAt: attribution.fetchStartedAt, + skippedAcceptances: 0, + renderCommitMs: null, + presentationFrameMs: null, + metricLabel: arm.metricLabel, + beforeMetricValue: arm.beforeMetricValue, + afterMetricValue: readMetric(arm.metricLabel), + expectedMetricValue: null, + metricChanged: null, + triggerAt: trigger.at, + triggerAcceptedSequence: trigger.acceptedSequence, + triggerRevision: trigger.revision + }; + } + // Find the (accepted model, rendered content) pair that belongs to THIS + // sample: an acceptance after the armed sequence whose render carries that + // revision's stamp AND the expected Home content for the change the harness + // triggered. A publication that does not move the Home metric (a + // provider-state-only revision, for example) cannot satisfy it, so an + // intermediate publication is skipped rather than mis-attributed. + let event = null; + let commit = null; + while (performance.now() < deadline && !commit) { + for (const candidate of acceptanceSink()) { + if (!candidate.revision || candidate.snapshotRevision !== candidate.revision || candidate.runtimeMapRevision !== candidate.revision || candidate.seq <= trigger.acceptedSequence) continue; +if (arm.expectedRevision && candidate.revision !== arm.expectedRevision) continue; +const rendered = bench.commits.find( +(entry) => +entry.at > candidate.at && +entry.textChanged && +entry.inStory && +entry.revision === candidate.revision && +// The seam-isolation control selects the first coherent acceptance after its +// boundary and its own rendered revision. It does not infer that this pair was +// caused by the fixture publication from timing, sequence, or visible content. +(arm.seamIsolation || entry.storyValue === arm.expectedMetricValue) + ); + if (rendered) { + event = candidate; + commit = rendered; + break; + } + } + if (!commit) await frame(); + } + if (!event || !commit) { + throw new Error( + "the Home content for the triggered change never rendered (expected " + + arm.metricLabel + + "=" + + arm.expectedMetricValue + + "; story=" + + JSON.stringify(bench.commits.filter((entry) => entry.inStory).slice(-4)) + + "; inStory commits=" + + bench.commits.filter((entry) => entry.inStory).length + + "; accepted=" + + JSON.stringify(acceptanceSink().slice(-3)) + + "; latest accepted seq=" + + trigger.acceptedSequence + + "->" + + (acceptanceSink().length ? acceptanceSink()[acceptanceSink().length - 1].seq : 0) + + "; notifications=" + + bench.events + + " opens=" + + bench.opens + + " latest notified=" + + bench.notifyRevision + + " " + + JSON.stringify(bench.notifyLog.slice(-3)) + + "; trigger=" + + JSON.stringify(trigger) + + "; arm=" + + JSON.stringify({ startedAt: arm.startedAt, previousSeq: arm.previousSeq, beforeMetricValue: arm.beforeMetricValue }) + + "; paired fetches=" + + JSON.stringify(bench.fetchLog.slice(-6)) + + ")" + ); + } + const acceptedAt = event.at; + const revision = event.revision; + // Stage 6 starts at the notification that caused the fetch cycle which + // delivered this accepted revision. See attributeNotification(). + const attribution = attributeNotification(revision, acceptedAt, trigger.at); + if (attribution.error) throw new Error(attribution.error + " (" + diagnostic() + ")"); + const notifyAt = attribution.notifyAt; + const skippedAcceptances = acceptanceSink().filter( + (entry) => entry.revision && entry.seq > trigger.acceptedSequence && entry.seq < event.seq + ).length; + const renderCommitAt = commit.at; + await frame(); + const presentedAt = performance.now(); + return { +notificationToCoherentModelMs: acceptedAt - notifyAt, +coherentModelToUsefulRenderMs: presentedAt - acceptedAt, +acceptanceAt: acceptedAt, +delayStartedAt: arm.seamIsolation ? Number(window.__dockermapBenchDelayStartedAt ?? NaN) : null, +acceptedRevision: revision, + snapshotRevision: event.snapshotRevision, + runtimeMapRevision: event.runtimeMapRevision, + acceptedSequence: event.seq, + notifiedRevision: attribution.notifiedRevision, + latestNotifiedRevision: bench.notifyRevision, + fetchDeliveredBy: attribution.deliveredBy, + fetchStartedAt: attribution.fetchStartedAt, + skippedAcceptances, + renderCommitMs: renderCommitAt - acceptedAt, + presentationFrameMs: presentedAt - renderCommitAt, + metricLabel: arm.metricLabel, + beforeMetricValue: arm.beforeMetricValue, + afterMetricValue: readMetric(arm.metricLabel), + expectedMetricValue: arm.expectedMetricValue, + metricChanged: arm.beforeMetricValue !== arm.expectedMetricValue, + triggerAt: trigger.at, + triggerAcceptedSequence: trigger.acceptedSequence, + triggerRevision: trigger.revision + }; + })(); + arm.task.catch((error) => { + arm.error = String(error && error.message ? error.message : error); + }); + bench.arm = arm; + return true; + }, + + armed() { + return Boolean(bench.arm && bench.arm.armed); + }, + + markModelPublicationTriggered() { + const arm = bench.arm; + if (!arm || !arm.task || !arm.armed) throw new Error("model acceptance was not armed"); + if (arm.trigger) throw new Error("the model publication trigger was already marked"); + const accepted = acceptanceSink().filter((entry) => entry.revision).slice(-1)[0] || null; + arm.trigger = { + at: performance.now(), + acceptedSequence: accepted ? accepted.seq : arm.previousSeq, + revision: accepted ? accepted.revision : "", + fetchLogLength: bench.fetchLog.length, + notifyLogLength: bench.notifyLog.length + }; + return arm.trigger; + }, + +setExpectedModelRevision(revision, delayMs) { + const arm = bench.arm; + if (!arm || !arm.task || !arm.armed) throw new Error("model acceptance was not armed"); + if (typeof revision !== "string" || revision.length === 0) { + throw new Error("the control publication did not supply a non-empty target model revision"); + } + arm.expectedRevision = revision; + // The controller acknowledgement proves causality; the real acceptance seam + // still has to observe this exact coherent pair before it can be measured. + window.__dockermapBenchRenderDelayTarget = revision; + window.__dockermapBenchRenderDelayMs = Number(delayMs) || 0; +return revision; +}, + +armSeamIsolationDelay(delayMs) { +const arm = bench.arm; +if (!arm || !arm.task || !arm.armed || !arm.seamIsolation) throw new Error("seam isolation was not armed"); +if (!Number.isFinite(Number(delayMs)) || Number(delayMs) <= 0) throw new Error("seam isolation requires a positive delay"); +// The application consumes this only after recordModelAcceptance() has +// recorded the next coherent pair. No publication identity is supplied. +window.__dockermapBenchRenderDelayMs = Number(delayMs); +window.__dockermapBenchDelayAfterNextAcceptance = true; +window.__dockermapBenchDelayStartedAt = undefined; +return true; +}, + + async awaitModelAcceptance() { + const arm = bench.arm; + if (!arm || !arm.task) throw new Error("model acceptance was not armed"); + try { + arm.result = await arm.task; + } catch (error) { + throw new Error(arm.error || String(error && error.message ? error.message : error)); + } finally { + arm.armed = false; + } + return arm.result; + }, + + async commandQuery(limit, preferredToken) { + const started = performance.now(); + const deadline = performance.now() + limit; + const palette = () => document.querySelector('[aria-label="Command palette"]'); + window.dispatchEvent(new KeyboardEvent("keydown", { key: "k", ctrlKey: true, bubbles: true })); + while (performance.now() < deadline && !palette()) { + await frame(); + } + const dialog = palette(); + if (!dialog) throw new Error("command palette did not open"); + const input = dialog.querySelector("input"); + if (!input) throw new Error("command palette has no query input"); + const listText = () => { + const items = dialog.querySelectorAll("li"); + return Array.from(items) + .map((item) => item.textContent || "") + .join("|"); + }; + // Snapshot the UNFILTERED list: the palette renders every command on open, + // so "some item exists" would pass even with filtering completely broken. + const unfiltered = listText(); + const unfilteredCount = dialog.querySelectorAll("li").length; + const tokens = unfiltered.match(/[A-Za-z0-9][A-Za-z0-9_.:-]{3,}/g) || []; + // A query with a known expected result: prefer the fixture-derived token + // the harness passes in; otherwise take a digit-bearing token from the + // rendered list itself. Either way the filtered list must still contain it. + const token = + preferredToken && tokens.includes(preferredToken) + ? preferredToken + : tokens.filter((value) => /\d/.test(value)).sort((left, right) => right.length - left.length)[0] || + tokens[0]; + if (!token) throw new Error("no queryable token exists in the unfiltered command list"); + const setter = Object.getOwnPropertyDescriptor(HTMLInputElement.prototype, "value").set; + setter.call(input, token); + input.dispatchEvent(new Event("input", { bubbles: true })); + while (performance.now() < deadline) { + const current = listText(); + // Success requires the filtered list to have actually narrowed — not just + // changed. The palette prepends an "Ask Copilot" item on any query, so a + // reorder-only or no-op filter would still change the joined text and + // still contain the token. The count must strictly drop. + if (current !== unfiltered && current.includes(token) && dialog.querySelectorAll("li").length < unfilteredCount) { + return performance.now() - started; + } + await frame(); + } + throw new Error("query " + token + " never produced a filtered list containing it"); + }, + + navigationDuration() { + const entry = performance.getEntriesByType("navigation")[0]; + return entry ? entry.duration : 0; + }, + + homeReady() { + const value = readMetric("Services"); + return value !== null && value !== ""; + }, + + currentRevision() { + return bench.notifyRevision || ""; + }, + + currentAcceptedRevision() { + const entries = acceptanceSink().filter((entry) => entry.revision); + return entries.length ? entries[entries.length - 1].revision : ""; + }, + + currentAcceptedSeq() { + const entries = acceptanceSink().filter((entry) => entry.revision); + return entries.length ? entries[entries.length - 1].seq : 0; + }, + + acceptedEventCount() { + return acceptanceSink().length; + }, + + /** Every origin the application fetched from (proof there is no daemon shortcut). */ + requestOrigins() { + return bench.requestOrigins.slice(); + }, + + /** The stream URL the application's real notification path opened. */ + streamUrl() { + return bench.streamUrl; + } + }; +})(); diff --git a/tests/perf/browserProbe.test.mjs b/tests/perf/browserProbe.test.mjs new file mode 100644 index 00000000..3e15895c --- /dev/null +++ b/tests/perf/browserProbe.test.mjs @@ -0,0 +1,40 @@ +/** Regression coverage for the Stage-6/7 seam-isolation boundary (#335). */ +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { resolve } from "node:path"; +import test from "node:test"; +import vm from "node:vm"; + +const probe = readFileSync(resolve(new URL(".", import.meta.url).pathname, "browserProbe.js"), "utf8"); + +function installProbe() { +let clock = 0; +const window = { fetch() {}, EventSource: function EventSource() {}, requestAnimationFrame: (done) => setImmediate(() => done()), performance: { now: () => ++clock } }; +window.EventSource.prototype = { addEventListener() {} }; +const document = { documentElement: { dataset: {}, querySelectorAll: () => [] }, querySelectorAll: () => [], addEventListener() {} }; +const context = { window, document, performance: window.performance, requestAnimationFrame: window.requestAnimationFrame, MutationObserver: class { observe() {} }, Element: class {}, URL, location: { href: "http://probe.test/" }, setImmediate, Promise, String, Number, Boolean, Array, JSON, Object, RegExp }; +vm.runInNewContext(probe, context); +window.__dockermapBenchAcceptanceSink = []; +return window; +} + +test("seam isolation selects the first coherent post-boundary pair without publication identity or content matching", async () => { +const window = installProbe(); +const helpers = window.__dockermapBenchHelpers; +helpers.armModelAcceptance({ mode: "content", seamIsolation: true, previousSeq: 0, limit: 1_000, metricLabel: "Offline" }); +helpers.armSeamIsolationDelay(250); +assert.equal(window.__dockermapBenchDelayAfterNextAcceptance, true); +window.__dockermapBench.notifyLog.push({ at: 1, revision: "unrelated" }); +window.__dockermapBench.fetchLog.push({ url: "snapshot", startedAt: 2, at: 3, revision: "unrelated" }); +window.__dockermapBenchAcceptanceSink.push({ seq: 1, at: 4, revision: "unrelated", snapshotRevision: "unrelated", runtimeMapRevision: "unrelated" }); +window.__dockermapBench.commits.push({ at: 5, inHome: true, inStory: true, textChanged: true, revision: "unrelated", storyValue: "not-a-fixture-sentinel" }); +const measured = await helpers.awaitModelAcceptance(); +assert.equal(measured.acceptedRevision, "unrelated"); +assert.equal(measured.acceptedSequence, 1); +assert.equal(measured.triggerRevision, ""); +}); + +test("the probe source keeps normal trigger identity separate from seam isolation", () => { +assert.match(probe, /armSeamIsolationDelay/); +assert.match(probe, /arm\.seamIsolation \|\| entry\.storyValue === arm\.expectedMetricValue/); +}); diff --git a/tests/perf/capture.ts b/tests/perf/capture.ts new file mode 100644 index 00000000..6cf5573a --- /dev/null +++ b/tests/perf/capture.ts @@ -0,0 +1,1942 @@ +#!/usr/bin/env node +/** + * DockerMap time-to-answer capture (#335) — the single documented command. + * + * npm run perf:time-to-answer -- \ + * --metadata /controlled/time-to-answer-metadata.json \ + * --output /controlled/time-to-answer-baseline.json \ + * [--baseline /controlled/previous-baseline.json] [--raw-dir /controlled/raw] + * + * It owns every test process: the deterministic fixture Docker daemon, the real + * daemon (with bench attribution), the real API, the production web build, the + * benchmark-only probe build, and real Chromium. + * + * It fails closed. A missing environment field, a failed process, a missing + * stage or an incomplete sample set aborts the capture instead of emitting a + * partial artifact; whatever raw samples were gathered are preserved next to + * the output for diagnosis. + * + * It never touches the live DockerMap deployment: all ports are reserved from + * the OS, every socket and directory is private to the run, and children are + * torn down by process group — never by pattern matching. + */ +import { spawn, spawnSync } from "node:child_process"; +import { createHash } from "node:crypto"; +import { appendFileSync, existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, writeFileSync } from "node:fs"; +import { request } from "node:http"; +import { tmpdir } from "node:os"; +import { join, dirname, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; +import { chromium } from "playwright"; +import { + TIME_TO_ANSWER_BASELINE, + TIME_TO_ANSWER_CONTROLLED_RUNS, + TIME_TO_ANSWER_END_TO_END_MATRIX, + TIME_TO_ANSWER_METHODOLOGY, + TIME_TO_ANSWER_REFERENCE_FIXTURES, + TIME_TO_ANSWER_STAGES, + TIME_TO_ANSWER_STAGE_KIND, + TIME_TO_ANSWER_WARMED_SAMPLES, + TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS, + TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS, + TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES, + TIME_TO_ANSWER_WARM_UP_METRICS, + assertDaemonBinaryProvenance, + assertTimeToAnswerEnvironment, + assertTimeToAnswerPromotion, + derivedTimeToAnswerPhaseNormalized, + deriveWarmUpCalibrationReport, + splitWarmedObservations, + TIME_TO_ANSWER_STATIONARITY_MAX_RATIO, + TIME_TO_ANSWER_STATIONARITY_MIN_RATIO +} from "../../apps/web/src/lib/performance/timeToAnswerEvidence"; +import { + POLL_PHASE_CONTROL_TOLERANCE_MS, + POLL_PHASE_DIVISIONS, + POLL_PHASE_MIN_SAMPLES_PER_PHASE, + assertFreeRunningPhaseSamples, + assertControlledPhaseEvidence, + assertPollPhaseSweep, + declaredPhaseForSample, + intendedLatencyMs, + isPhaseControlledFixture, + observedPhaseBucketMs, + phaseMediansMs, + pollPhaseGridMs, + type PollPhaseSweep +} from "../../apps/web/src/lib/performance/timeToAnswerPollPhase"; +import { FIXTURE_REVISION, SLOW_COMPOSE_SERVICES, buildSlowComposeProject, expectedExitedCount } from "./dockerFixtureTopology.mjs"; +import { reservePort, startStaticServer } from "./staticServer.mjs"; +import { withFreshBrowserRuns } from "./browserLifecycle.mjs"; +import { startStageFivePublicationController } from "./stageFivePublicationControl.mjs"; +import { armCaptureStageFivePublication } from "./stageFiveCaptureControl.mjs"; + +const REPO_ROOT = resolve(fileURLToPath(new URL("../..", import.meta.url))); +const sleep = (ms: number) => new Promise((done) => setTimeout(done, ms)); +const nowMs = () => Number(process.hrtime.bigint()) / 1e6; + +type RawSamples = Record>; +type ProbeMeasurement = { buildModelMs: number[]; legacyTopologyLayoutMs: number[] }; + +function parseArgs(argv: string[]): Record { + return Object.fromEntries( + argv.flatMap((value, index, all) => + value.startsWith("--") ? [[value.slice(2), all[index + 1] ?? ""]] : [] + ) + ); +} + +const args = parseArgs(process.argv.slice(2)); +const metadataPath = args.metadata; +const outputPath = args.output; +const baselinePath = args.baseline; +const rawDir = args["raw-dir"]; +const calibrationOutputPath = args["calibration-output"]; +const calibration = Boolean(calibrationOutputPath); +const runs = Number(args.runs ?? (calibration ? 1 : TIME_TO_ANSWER_CONTROLLED_RUNS)); +const samples = Number(args.samples ?? (calibration ? TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS : TIME_TO_ANSWER_WARMED_SAMPLES)); +const onlyFixtures = args.fixtures ? args.fixtures.split(",").map((name) => name.trim()) : null; +if (onlyFixtures && process.env.DOCKERMAP_BENCH_DEBUG !== "1") { + throw new Error( + "--fixtures only exists for probing individual fixtures during development; a partial run can never satisfy the closed matrix, so gate it behind DOCKERMAP_BENCH_DEBUG=1." + ); +} +/** + * Fixture-derived Cmd-K query token: container 0 always carries this name. + * The poll interval is derived after the environment is parsed, below. + */ +const FixtureTopologyQueryToken = "fixture-service-0"; + +/** + * The Home metric the stage-7 "expected content" check asserts. The fixture's + * generation delta changes exactly this metric (generation `g` stops the first + * `g` containers), so the check is discriminating: a stale render — or a render + * that belongs to a different revision — cannot satisfy it. + */ +const HomeMetricLabel = "Offline"; + +if (!metadataPath || (!outputPath && !calibrationOutputPath)) { + throw new Error( + "Usage: npm run perf:time-to-answer -- --metadata --output " + ); +} +// Diagnostics are deliberately outside the closed artifact and all measurement +// calculations. A failed append must never affect a capture result. +const diagnosticDirectory = process.env.DOCKERMAP_BENCH_DIAG_DIR || rawDir || dirname(outputPath ?? calibrationOutputPath!); +function appendDiagnostic(file: "layers.jsonl" | "lifecycle.jsonl" | "fixture-identity.jsonl", record: Record): void { + try { + mkdirSync(diagnosticDirectory, { recursive: true }); + appendFileSync(join(diagnosticDirectory, file), `${JSON.stringify(record)}\n`); + } catch { + // Diagnostics are observational only and must not throw into measurement. + } +} +function recordLifecycle(event: "browser_launch" | "context_create" | "page_create" | "navigate" | "teardown_start" | "teardown_end" | "exception" | "closure", step?: string, error?: unknown): void { + appendDiagnostic("lifecycle.jsonl", { + event, + step: step ?? null, + error_text: error === undefined ? null : String(error), + monotonic_timestamp: nowMs() + }); +} +async function drainLayerDiagnostics(page: any, fixture: string): Promise { + try { + const records = await page.evaluate("window.__dockermapBenchLayerSink ? window.__dockermapBenchLayerSink.splice(0) : []"); + if (Array.isArray(records)) for (const record of records) appendDiagnostic("layers.jsonl", { ...record, fixture }); + } catch (error) { + recordLifecycle("exception", "drain_layer_diagnostics", error); + } +} +if (!calibration && ( + (runs !== TIME_TO_ANSWER_CONTROLLED_RUNS || samples !== TIME_TO_ANSWER_WARMED_SAMPLES) && + process.env.DOCKERMAP_BENCH_DEBUG !== "1" +)) { + throw new Error( + `The contract requires exactly ${TIME_TO_ANSWER_CONTROLLED_RUNS} controlled runs and ${TIME_TO_ANSWER_WARMED_SAMPLES} warmed samples per cell.` + ); +} + +const metadata = JSON.parse(readFileSync(metadataPath, "utf8")) as { + environment: unknown; + daemonBinary?: string; +}; +assertTimeToAnswerEnvironment(metadata.environment); +const environment = metadata.environment; + +/** + * The effective SSE poll interval. It is passed to the API explicitly so the + * recorded pin cannot drift from the interval that ran, and it defines the stage-5 + * phase grid the harness drives: the publication phase relative to the observation + * stream's poll ticks is chosen from a declared grid over this interval, never left + * to a random trigger delay. + */ +const pollIntervalMs = Number(environment.ssePollIntervalMs); +if (!Number.isFinite(pollIntervalMs) || pollIntervalMs <= 0) { + throw new Error("the pinned ssePollIntervalMs must be a positive number"); +} + +// A baseline is only reproducible if the code that measured it is committed. +// The refusal is deliberate: an uncommitted harness produces numbers nobody can +// re-derive, which is exactly how baseline 1 was invalidated in review. +function gitOutput(gitArgs: string[]): string { + return run("git", gitArgs).trim(); +} +const dirtyTree = gitOutput(["status", "--porcelain"]); +if (dirtyTree !== "") { + throw new Error( + `refusing to capture from a dirty worktree — commit the harness first so the baseline is reproducible:\n${dirtyTree}` + ); +} +const productRevision = gitOutput(["rev-parse", "HEAD"]); +if (environment.sourceRevision !== productRevision) { + throw new Error( + `metadata sourceRevision (${environment.sourceRevision}) is not the checked-out commit (${productRevision}); re-emit the metadata so the artifact names the revision it actually measured` + ); +} +const harnessRevision = gitOutput(["log", "-1", "--format=%H", "--", "tests/perf", "apps/web/src/lib/performance"]); +if (environment.harnessRevision !== harnessRevision) { + throw new Error( + `metadata harnessRevision (${environment.harnessRevision}) is not the harness commit (${harnessRevision})` + ); +} +// The methodology is part of the measurement, not metadata trivia: a candidate is +// only comparable against a baseline captured under the same design (stage-5 phase +// control, fixed warm-up protocol, stationarity guard, provenance/compatibility +// split). A stale metadata file must not silently capture under the old design. +if (environment.methodologyVersion !== TIME_TO_ANSWER_METHODOLOGY) { + throw new Error( + `metadata methodologyVersion (${environment.methodologyVersion}) is not the contract's ` + + `(${TIME_TO_ANSWER_METHODOLOGY}); re-emit the metadata so the artifact names the design it measured` + ); +} +const daemonBinary = metadata.daemonBinary ?? join(REPO_ROOT, "crates/target/release/dockermap-daemon"); +// Bind the executed binary to the recorded revision before and after the run: +// stages 1/2/3/4/5/9 all come from this executable, so a stale or substituted +// binary would misattribute every daemon-side number. +function currentDaemonSha256(): string { + return createHash("sha256").update(readFileSync(daemonBinary)).digest("hex"); +} +assertDaemonBinaryProvenance({ + expectedSha256: environment.daemonBinarySha256, + observedSha256: currentDaemonSha256(), + phase: "before capture" +}); +const daemonBinarySha256Before = currentDaemonSha256(); +const launchArgs = (environment.browserFlags as string[]).filter(Boolean); + +const plans = TIME_TO_ANSWER_REFERENCE_FIXTURES.filter( + (fixture) => !onlyFixtures || onlyFixtures.includes(fixture.name) +).filter((fixture) => !calibration || fixture.kind === "reference").map((fixture) => ({ + name: fixture.name, + containers: fixture.containers, + scenario: fixture.kind === "reference" ? "reference" : fixture.name +})); + +const raw: RawSamples = {}; +/** + * Stage-5 poll-phase sweep records (methodology revision 2). Harness-only: they + * carry the declared and observed phase of every recorded stage-5 sample, and are + * written beside the artifact rather than inside it, so the closed evidence + * schema stays raw numbers only. + */ +const stageFiveSweep: Array = []; +/** Per-fixture result of the declared phase-sweep validity guards. */ +const stageFiveValidity: Record = {}; +/** + * The complete warmed observation window per `fixture|stage`, in the order the + * daemon produced it (`samples + TIME_TO_ANSWER_WARM_UP_OBSERVATIONS` values). + * The fixed warm-ups occupy the declared leading indices, so the retention rule is verifiable from + * the raw series instead of being asserted only by the code that applied it. + */ +const warmedObservationWindows: Record = {}; +const MATRIX = new Set(TIME_TO_ANSWER_END_TO_END_MATRIX.map((cell) => `${cell.fixture}|${cell.stage}`)); +/** A cell only exists if the closed contract declares it for this fixture. */ +const hasStage = (fixture: string, stage: string) => MATRIX.has(`${fixture}|${stage}`); +function record(fixture: string, stage: string, values: number[]): void { + if (!hasStage(fixture, stage)) return; + // Fail at the measurement site, naming the cell, rather than at artifact + // validation where the origin is no longer recoverable. + for (const value of values) { + if (typeof value !== "number" || !Number.isFinite(value) || value < 0) { + throw new Error(`non-numeric ${stage} sample for ${fixture}: ${JSON.stringify(value)}`); + } + } + if (values.length === 0) { + throw new Error(`no samples were measured for ${stage} on ${fixture}`); + } + // `values` is one complete controlled run: its samples, in order. + raw[fixture]![stage]!.push(values); +} + function recordWarmed(fixture: string, stage: string, values: number[], run: number): void { + if (!hasStage(fixture, stage)) return; + if (calibration) { record(fixture, stage, values); return; } + const { warmUps: burnIns, recorded } = splitWarmedObservations(values, samples); + const label = `${fixture}|${stage}|run${run}`; + burnInObservations[label] = burnIns; + warmedObservationWindows[label] = values.slice(0, TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS + samples); + record(fixture, stage, recorded); + } +for (const plan of plans) { + raw[plan.name] = {}; + for (const stage of TIME_TO_ANSWER_STAGES) { + if (hasStage(plan.name, stage.id)) raw[plan.name]![stage.id] = []; + } +} + +function preserveRaw(reason: string): void { + const destination = rawDir + ? join(rawDir, "time-to-answer-raw.json") + : `${outputPath ?? calibrationOutputPath}.raw.json`; + try { + mkdirSync(dirname(destination), { recursive: true }); + writeFileSync( + destination, + JSON.stringify( + { + reason, + raw, + warmedObservationWindows, + burnInObservations + }, + null, + 2 + ) + ); + process.stderr.write(`[capture] preserved raw samples at ${destination}\n`); + } catch (error) { + process.stderr.write(`[capture] could not preserve raw samples: ${String(error)}\n`); + } +} + +function run(command: string, commandArgs: string[], env: NodeJS.ProcessEnv = {}) { + const result = spawnSync(command, commandArgs, { + cwd: REPO_ROOT, + env: { ...process.env, ...env }, + encoding: "utf8", + maxBuffer: 64 * 1024 * 1024 + }); + if (result.status !== 0) { + throw new Error(`${command} ${commandArgs.join(" ")} failed:\n${result.stdout}\n${result.stderr}`); + } + return result.stdout; +} + +function spawnOwned(command: string, commandArgs: string[], env: NodeJS.ProcessEnv = {}) { + const child = spawn(command, commandArgs, { + cwd: REPO_ROOT, + env: { ...process.env, ...env }, + stdio: ["ignore", "pipe", "pipe"], + detached: true // own process group: teardown kills the whole tree + }); + child.stdout?.resume(); + child.stderr?.resume(); + return child; +} + +/** + * Start a child that binds a reserved TCP port and waits until it answers. + * + * reservePort() cannot be race-free (it binds, closes, and the port is handed + * back to the pool), so an unrelated process — or one of our own children not yet + * reaped — can take the port before the child binds. A collision used to abort an + * expensive capture; here it is retried on a fresh port, bounded, and a genuinely + * broken child still fails the capture after the attempts are exhausted. + */ +async function startChildOnFreePort(input: { + name: string; + attempts?: number; + /** Keep the same port across attempts (required when the port is baked elsewhere). */ + fixedPort?: number; + spawnOn: (port: number) => any; + ping: (port: number) => Promise; +}): Promise<{ child: any; port: number }> { + const attempts = input.attempts ?? 4; + let lastError: unknown = null; + for (let attempt = 1; attempt <= attempts; attempt += 1) { + const port = input.fixedPort ?? (await reservePort()); + const child = input.spawnOn(port); + const deadline = Date.now() + 60_000; + let ready = false; + while (Date.now() < deadline && !ready) { + if (child.exitCode !== null || child.signalCode) break; + ready = await input.ping(port); + if (!ready) await sleep(25); + } + if (ready) return { child, port }; + stopOwned(child); + lastError = new Error(`${input.name} did not become ready on port ${port}`); + process.stderr.write( + `[capture] ${input.name} failed to bind port ${port}; retrying (attempt ${attempt}/${attempts})\n` + ); + await sleep(250); + } + throw lastError ?? new Error(`${input.name} never started`); +} + +function stopOwned(child: { pid?: number; exitCode: number | null; signalCode?: NodeJS.Signals | null } | null) { + if (!child?.pid) return; + if (child.exitCode !== null || child.signalCode) return; + try { + process.kill(-child.pid, "SIGKILL"); + } catch { + try { + process.kill(child.pid, "SIGKILL"); + } catch { + // already gone + } + } +} + +async function fetchJson(url: string, timeoutMs = 5_000): Promise { + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(), timeoutMs); + try { + const response = await fetch(url, { signal: controller.signal }); + if (!response.ok) return null; + return await response.json(); + } catch { + return null; + } finally { + clearTimeout(timer); + } +} + +async function waitForJson(url: string, predicate: (value: any) => boolean, timeoutMs: number) { + const deadline = Date.now() + timeoutMs; + while (Date.now() < deadline) { + const value = await fetchJson(url, 2_000); + if (value && predicate(value)) return value; + await sleep(4); + } + throw new Error(`timed out waiting for ${url}`); +} + +/** + * The seam's in-page sink identifier. Its presence is what proves an artifact + * carries the acceptance seam; its absence is what proves the shipped product + * does not. + */ +const SEAM_SINK_IDENTIFIER = "__dockermapBenchAcceptanceSink"; + +function containsSeamIdentifier(directory: string): boolean { + let found = false; + const visit = (dir: string): void => { + for (const entry of readdirSync(dir, { withFileTypes: true })) { + const full = join(dir, entry.name); + if (entry.isDirectory()) visit(full); + else if ( + /\.(js|mjs|html)$/.test(entry.name) && + readFileSync(full, "utf8").includes(SEAM_SINK_IDENTIFIER) + ) { + found = true; + } + } + }; + visit(directory); + return found; +} + +/** + * Bind THIS capture to the artifacts it will serve, before anything is measured: + * the benchmark-mode application build must carry the acceptance seam (otherwise + * stage 6 would be measuring something other than the application's acceptance) + * and the shipped production build must not. The same property is checked in CI + * by tests/perf/productionIsolation.test.mjs; here it guards the running capture. + */ +function assertBuildIsolation(): void { + const product = join(REPO_ROOT, "apps/web/dist"); + const benchmark = join(REPO_ROOT, "tests/perf/.bench-app-dist"); + if (containsSeamIdentifier(product)) { + throw new Error( + "the production web build contains the benchmark acceptance seam: refusing to capture from a build whose numbers would not describe the shipped product" + ); + } + if (!containsSeamIdentifier(benchmark)) { + throw new Error( + "the benchmark-mode application build does not contain the acceptance seam: stage 6 would not be the application's coherent-model acceptance" + ); + } +} + +/** + * Advance the fixture's topology generation and CONFIRM it landed. The read-back is + * the point: a trigger that silently did not reach the fixture daemon would leave the + * publication grid and the stage-7 expectation describing a change that never + * happened, and the stage-7 check would then blame the application for it. + */ +async function setFixtureGeneration(socketPath: string, generation: number): Promise> { + await postUnix(socketPath, `/__fixture/topology-generation/${generation}`); + const state = JSON.parse(await getUnix(socketPath, "/__fixture/state")) as Record & { generation?: number }; + if (state.generation !== generation) { + throw new Error( + `the fixture trigger did not land: asked for generation ${generation}, fixture reports ${String(state.generation)}` + ); + } + return state; +} + +function getUnix(socketPath: string, path: string): Promise { + return new Promise((done, fail) => { + const call = request({ socketPath, path, method: "GET" }, (response) => { + let body = ""; + response.on("data", (chunk) => (body += chunk)); + response.on("end", () => + response.statusCode === 200 ? done(body) : fail(new Error(`fixture read ${response.statusCode}`)) + ); + }); + call.on("error", fail); + call.end(); + }); +} + +/** Count the containers a fixture or daemon payload reports as exited/offline. */ +function countExited(body: string): number { + try { + const parsed = JSON.parse(body) as { containers?: Array<{ state?: string; status?: string; State?: string }> }; + const containers = parsed.containers ?? (parsed as unknown as Array<{ state?: string; status?: string; State?: string }>); + if (!Array.isArray(containers)) return -1; + return containers.filter((container) => { + const state = String(container.state ?? container.State ?? "").toLowerCase(); + const status = String(container.status ?? "").toLowerCase(); + return state === "offline" || state === "exited" || status.includes("exited"); + }).length; + } catch { + return -1; + } +} + +/** POST to a unix-socket HTTP endpoint (fixture daemon control route). */ +function postUnix(socketPath: string, path: string): Promise { + return new Promise((done, fail) => { + const call = request({ socketPath, path, method: "POST", headers: { "content-length": 0 } }, (response) => { + let body = ""; + response.on("data", (chunk) => (body += chunk)); + response.on("end", () => (response.statusCode === 200 ? done(body) : fail(new Error(`fixture control ${response.statusCode}`)))); + }); + call.on("error", fail); + call.end(); + }); +} + +function writeComposeProject(root: string, scenario: string): void { + mkdirSync(root, { recursive: true }); + if (scenario === "slow-bounded-compose-projection") { + writeFileSync(join(root, "compose.yaml"), buildSlowComposeProject()); + return; + } + const lines = ["name: dockermap-fixture", "services:"]; + for (let index = 0; index < 40; index += 1) { + lines.push(` fixture-service-${index}:`); + lines.push(" image: dockermap/fixture:1"); + lines.push(" volumes:"); + lines.push(` - ./fixture-data-${index}:/data`); + } + writeFileSync(join(root, "compose.yaml"), `${lines.join("\n")}\n`); +} + +const BENCH_STAGE_KEYS = ["dockerObservationMs", "composeEnrichmentMs", "findingsDerivationMs"] as const; +/** + * The daemon's first-ever observation runs before its listener binds, so its first + * passes are a cold start. For warmed stages a FIXED number of warm-up + * observations (`TIME_TO_ANSWER_WARM_UP_OBSERVATIONS`, declared before the capture) + * are excluded from recorded samples and retained here instead, so conditioning is + * auditable rather than silent and never chosen from the data. + */ +const burnInObservations: Record = {}; + +function readBenchSink(path: string): Record<(typeof BENCH_STAGE_KEYS)[number], number[]> { + const stages = { dockerObservationMs: [], composeEnrichmentMs: [], findingsDerivationMs: [] } as Record< + (typeof BENCH_STAGE_KEYS)[number], + number[] + >; + if (!existsSync(path)) return stages; + for (const line of readFileSync(path, "utf8").split("\n")) { + if (!line.trim()) continue; + try { + const record = JSON.parse(line) as { stage?: string; ms?: number }; + const key = record.stage as (typeof BENCH_STAGE_KEYS)[number]; + if (key && stages[key] && Number.isFinite(record.ms)) stages[key].push(record.ms as number); + } catch { + // ignore a partially written trailing line + } + } + return stages; +} + +async function waitForBenchSamples(path: string, count: number, timeoutMs: number) { + const deadline = Date.now() + timeoutMs; + while (Date.now() < deadline) { + const stages = readBenchSink(path); + if (BENCH_STAGE_KEYS.every((key) => stages[key].length >= count)) { + return { + dockerObservationMs: stages.dockerObservationMs.slice(0, count), + composeEnrichmentMs: stages.composeEnrichmentMs.slice(0, count), + findingsDerivationMs: stages.findingsDerivationMs.slice(0, count) + }; + } + await sleep(50); + } + throw new Error("daemon bench attribution did not accumulate the required warmed samples"); +} + +/** + * Stage 5 — daemon publication committed -> the Node/SSE layer observes the new + * revision through TODAY'S real polling mechanism, poll wait included. + * + * Methodology revision 2 drives the phase DETERMINISTICALLY instead of sleeping a + * uniform random delay and hoping the samples land across the interval. Baseline 3 + * disproved that hope: reference-25's 45 samples sat in a 120 ms band (6.0 % of the + * interval) because both the daemon's refresh loop and the API's poller are fixed + * 2 s loops, so the measured gap was their phase offset, not a sample of any + * distribution. The design is declared in `timeToAnswerPollPhase.ts`: a fixed grid + * of ten phases spanning the interval. Each 15-sample run sweeps the grid + * ascending and repeats phases 0–4; every sample's phase is recoverable from its + * position in its run, every phase has at least two observations, and phase + * normalization weights each phase equally rather than weighting repeats more. + * + * The harness arms a benchmark-only fixture controller before triggering its + * generation. The controller observes an actual API health poll, releases the + * exact armed revision at the declared offset, and acknowledges that identity. + * Each sample verifies that exact revision through the real API stream; no daemon + * grid prediction is used. + */ +const PHASE_CONNECT_MARGIN_MS = 30; + +interface PublicationTracker { + /** Exact publication instants of the daemon's refresh cycles, in order. */ + readonly cycles: number[]; + /** Every revision change the daemon published, for the chain audit. */ + readonly revisionChanges: Array<{ at: number; revision: string }>; + /** The revision the daemon currently publishes. */ + revision(): string; + waitForCycles(count: number, timeoutMs: number): Promise; + /** Least-squares fit of the cycle grid: the daemon's refresh period. */ + periodMs(): number; + lastCycleAtMs(): number; + /** The cycle boundary a poll tick carried: the last one at or before it. */ + boundaryAtOrBefore(instant: number): number | null; + waitForRevisionAfter(previousRevision: string, afterMs: number, timeoutMs: number): Promise<{ at: number; revision: string }>; + stop(): Promise; +} + +/** + * Tracks the daemon's publication GRID from the snapshot's `lastUpdated`, which the + * daemon stamps once per refresh cycle (epoch milliseconds), rather than from + * revision changes: one refresh cycle can publish a provider-state revision a few + * hundred milliseconds after the Docker snapshot, and fitting the grid over every + * revision change skewed the period by hundreds of milliseconds — which stage 5's + * deepest declared phases amplify one-for-one into phase error. + * + * Using the stamp also removes detection lag: the publication instant is the + * daemon's own timestamp, aligned once to the harness's monotonic clock. + */ +async function startPublicationTracker( + daemonPort: number, + initialRevision: string, + _intervalMs: number +): Promise { + const url = `http://127.0.0.1:${daemonPort}/daemon/health`; + const cycles: number[] = []; + const revisionChanges: Array<{ at: number; revision: string }> = []; + let offsetMs: number | null = null; + let revision = initialRevision; + let lastUpdated = 0; + let stopped = false; + const running = (async () => { + while (!stopped) { + const health = await fetchJson(url, 1_000); + const stamp = Number((health as { lastUpdated?: number } | null)?.lastUpdated ?? 0); + const next = (health?.modelRevision as string | undefined) ?? ""; + if (stamp > 0) { + if (offsetMs === null) offsetMs = nowMs() - stamp; + if (stamp !== lastUpdated) { + const publishedAt = stamp + offsetMs; + lastUpdated = stamp; + cycles.push(publishedAt); + if (next && next !== revision) { + revision = next; + revisionChanges.push({ at: publishedAt, revision: next }); + } + } else if (next && next !== revision) { + // A revision the daemon published without advancing the snapshot stamp: + // recorded for the chain audit, never used as a grid boundary. + revision = next; + revisionChanges.push({ at: nowMs(), revision: next }); + } + } + await sleep(10); + } + })(); + return { + cycles, + revisionChanges, + revision: () => revision, + async waitForCycles(count: number, timeoutMs: number) { + const deadline = Date.now() + timeoutMs; + while (cycles.length < count && Date.now() < deadline) await sleep(10); + if (cycles.length < count) { + throw new Error( + `the daemon published only ${cycles.length} refresh cycles; stage 5 needs ${count} to know its publication grid` + ); + } + }, + periodMs() { + const recent = cycles.slice(-8); + if (recent.length < 2) { + throw new Error("stage 5 needs at least two observed publication cycles before it can predict the next one"); + } + const count = recent.length; + const meanIndex = (count - 1) / 2; + const meanAt = recent.reduce((sum, value) => sum + value, 0) / count; + let numerator = 0; + let denominator = 0; + recent.forEach((at, index) => { + numerator += (index - meanIndex) * (at - meanAt); + denominator += (index - meanIndex) ** 2; + }); + return numerator / denominator; + }, + lastCycleAtMs() { + const last = cycles[cycles.length - 1]; + if (last === undefined) throw new Error("stage 5 has observed no publication cycle yet"); + return last; + }, + boundaryAtOrBefore(instant: number) { + let latest: number | null = null; + for (const candidate of cycles) { + if (candidate <= instant) latest = candidate; + } + return latest; + }, + async waitForRevisionAfter(previousRevision: string, afterMs: number, timeoutMs: number) { + const deadline = Date.now() + timeoutMs; + while (Date.now() < deadline) { + const publication = revisionChanges.find((change) => change.revision !== previousRevision && change.at >= afterMs); + if (publication) return publication; + await sleep(2); + } + throw new Error("stage 5 could not witness the triggered daemon publication; the cell is invalidated"); + }, + async stop() { + stopped = true; + await running; + } + }; +} + +interface PhaseSamplePlan { + declaredPhaseMs: number; + intendedLatencyMs: number; + predictedPublicationAtMs: number; + connectedAtMs: number; +} + +/** + * Measure one stage-5 sample at its declared phase. `onConnected` runs after the + * observation stream is connected and before the publication is awaited, which is + * where the caller arms the browser measurement and triggers the fixture change: + * both must happen after the connection (so the poll tick carries the change) and + * before the publication (so the change is in it). + */ +async function observeStageFiveSample(input: { + tracker: PublicationTracker; + daemonPort: number; + apiPort: number; + webOrigin: string; + intervalMs: number; + runIndex: number; + sampleIndex: number; +mode: "phase-controlled" | "free-running"; + controllerUrl?: string; + onConnected: (plan: PhaseSamplePlan) => Promise<{ triggeredAtMs?: number }>; + timeoutMs?: number; +}): Promise<{ sample: PollPhaseSweep; revisions: string[]; tickCarriedNewerRevision: boolean }> { + const phaseControlled = input.mode === "phase-controlled"; + const declaredPhaseMs = declaredPhaseForSample(input.runIndex, input.sampleIndex, input.intervalMs); + const intended = intendedLatencyMs(declaredPhaseMs, input.intervalMs); + let predicted = 0; + // Free-running cells cannot place the publication (their revisions advance from the + // daemon's own host provider collection), so the observation is started now and the + // phase the sample achieves is recorded rather than driven. + // + // The revision that was current before the sample is read FRESH rather than taken + // from the tracker's last poll: the API emits the current revision immediately on + // connect, so a stale local copy makes that connect frame look like the sample's + // observation. + const previousRevision = + ( + (await fetchJson(`http://127.0.0.1:${input.daemonPort}/daemon/health`, 1_000)) as { + modelRevision?: string; + } | null + )?.modelRevision ?? input.tracker.revision(); + const triggerId = `stage-five-r${input.runIndex}-s${input.sampleIndex}-${Date.now()}`; +if (phaseControlled) { +if (!input.controllerUrl) throw new Error("controlled stage-5 sample has no publication controller"); +} +const publication = phaseControlled +? await armCaptureStageFivePublication({ +controllerUrl: input.controllerUrl!, +triggerId, +requestedPhaseMs: declaredPhaseMs, +previousRevision, +timeoutMs: input.timeoutMs ?? 45_000 +}) +: null; + const connectedAtMs = nowMs(); + // Controlled publications are held by the fixture controller until a real API + // poll has occurred; free-running cells retain their connection-frame guard. + const minimumObservationAtMs = phaseControlled ? connectedAtMs : connectedAtMs + PHASE_CONNECT_MARGIN_MS; + let ignoredEarlyFrames = 0; + + // The real API stream owns the unchanged 2000 ms cadence. The fixture controller, + // not a predicted daemon grid, releases an armed publication relative to its poll. + const controller = new AbortController(); + const observed = { at: 0, revisions: [] as string[] }; + const response = await fetch(`http://127.0.0.1:${input.apiPort}/api/events/stream`, { + headers: { accept: "text/event-stream", origin: input.webOrigin }, + signal: controller.signal + }); + if (!response.ok || !response.body) { + controller.abort(); + throw new Error(`stage 5 observation stream did not open (HTTP ${response.status})`); + } + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + const reading = (async () => { + let buffer = ""; + try { + for (;;) { + const { value, done } = await reader.read(); + if (done) return; + buffer += decoder.decode(value, { stream: true }); + const frames = buffer.split("\n\n"); + buffer = frames.pop() ?? ""; + for (const frame of frames) { + const dataLine = frame.split("\n").find((line) => line.startsWith("data:")); + if (!dataLine) continue; + try { + const payload = JSON.parse(dataLine.slice(5).trim()) as { modelRevision?: string }; + const revision = payload.modelRevision; + if (revision && revision !== previousRevision) { + if (!observed.at && nowMs() < minimumObservationAtMs) { + ignoredEarlyFrames += 1; + continue; + } + if (!observed.revisions.includes(revision)) observed.revisions.push(revision); + if (!observed.at) observed.at = nowMs(); + } + } catch { + // keepalive or non-JSON frame + } + } + } + } catch { + // stream closed by teardown + } + })(); + + try { +const trigger = () => input.onConnected({ +declaredPhaseMs, +intendedLatencyMs: intended, +predictedPublicationAtMs: predicted, +connectedAtMs +}); +if (phaseControlled) { +const actualPublication = await publication!.release(trigger); +// The exact release acknowledgement is resolved by the fixture controller, +// while the stream remains the only Stage-5 observation path. +const deadline = Date.now() + (input.timeoutMs ?? 45_000); +while (Date.now() < deadline && !observed.at) await sleep(2); +if (!observed.at) { +throw new Error( +`stage 5 did not observe a new revision at declared phase ${declaredPhaseMs.toFixed(1)} ms ` + +`(predicted publication ${(predicted - connectedAtMs).toFixed(1)} ms after connection)` +); +} +const publicationAtMs = actualPublication.releasedAtMs; +const observedLatencyMs = Math.max(0, observed.at - publicationAtMs); +const observedPhaseMs = actualPublication.releasedAtMs - actualPublication.pollAtMs; +if (actualPublication.revision !== observed.revisions[0]) { +throw new Error("stage 5 observed a revision other than the triggered publication; the cell is invalidated"); +} +if (Math.abs(observedPhaseMs - declaredPhaseMs) > POLL_PHASE_CONTROL_TOLERANCE_MS) { +throw new Error( +`stage 5 actual publication was ${(observedPhaseMs - declaredPhaseMs).toFixed(1)} ms from its intended phase; ` + +"phase control could not be established and the cell is invalidated" +); +} +const phaseErrorMs = observedLatencyMs - intended; +assertControlledPhaseEvidence({ +intendedPhaseMs: declaredPhaseMs, +observedPhaseMs, +publicationLatencyMs: observedLatencyMs, +phaseErrorMs +}); +const tickCarriedNewerRevision = input.tracker.revisionChanges.some( +(change) => change.at > publicationAtMs && change.at <= observed.at +); +return { +revisions: observed.revisions, +tickCarriedNewerRevision, +sample: { +runIndex: input.runIndex, +sampleIndex: input.sampleIndex, +declaredPhaseMs, +intendedPhaseMs: declaredPhaseMs, +observedPhaseMs, +intendedLatencyMs: intended, +connectedAtMs, +predictedPublicationAtMs: predicted, +observedPublicationAtMs: publicationAtMs, +observedObservationAtMs: observed.at, +observedLatencyMs, +publicationLatencyMs: observedLatencyMs, +observedPhaseBucketMs: observedPhaseBucketMs(observedLatencyMs, input.intervalMs), +phaseErrorMs, +observedVia: "api-sse", +observedRevision: observed.revisions[0] ?? "", +previousRevision, +phaseControlled: true +} +}; +} + +await trigger(); + +// The exact release acknowledgement is resolved by the fixture controller, while + // the stream remains the only Stage-5 observation path. + const deadline = Date.now() + (input.timeoutMs ?? 45_000); + while (Date.now() < deadline && !observed.at) await sleep(2); + if (!observed.at) { + throw new Error( + `stage 5 did not observe a new revision at declared phase ${declaredPhaseMs.toFixed(1)} ms ` + + `(predicted publication ${(predicted - connectedAtMs).toFixed(1)} ms after connection)` + ); + } + // Controlled cells require the exact acknowledged trigger revision; no later or + // nearest revision may be substituted. +const publicationAt = input.tracker.boundaryAtOrBefore(observed.at); + if (publicationAt === null) { + throw new Error("stage 5 lost the publication instant for this sample"); + } + const publicationAtMs = publicationAt; + const observedLatencyMs = Math.max(0, observed.at - publicationAtMs); +const observedPhaseMs = publicationAtMs - connectedAtMs; + // Recorded for the audit: whether the tick carried a revision published AFTER the + // boundary (an intra-cycle provider publication). The measurement stays the + // boundary's, because the declared stage-5 question is the publication the harness + // triggered, not the newest bytes the API happened to hold. + const tickCarriedNewerRevision = input.tracker.revisionChanges.some( + (change) => change.at > publicationAtMs && change.at <= observed.at + ); + // Free-running samples record the phase they ACHIEVED: the declared phase is the + // bucket the observation landed in, so the sample cannot claim a phase it did not + // drive. Controlled samples keep the declared phase they were driven to. +const recordedPhaseMs = input.intervalMs - observedPhaseBucketMs(observedLatencyMs, input.intervalMs); +const recordedIntended = input.intervalMs - recordedPhaseMs; +const phaseErrorMs = observedLatencyMs - recordedIntended; + return { + revisions: observed.revisions, + tickCarriedNewerRevision, + sample: { + runIndex: input.runIndex, + sampleIndex: input.sampleIndex, + declaredPhaseMs: recordedPhaseMs, +intendedPhaseMs: recordedPhaseMs, + observedPhaseMs, + intendedLatencyMs: recordedIntended, + connectedAtMs, +predictedPublicationAtMs: publicationAtMs, + observedPublicationAtMs: publicationAtMs, + observedObservationAtMs: observed.at, + observedLatencyMs, + publicationLatencyMs: observedLatencyMs, + observedPhaseBucketMs: observedPhaseBucketMs(observedLatencyMs, input.intervalMs), + phaseErrorMs, + observedVia: "api-sse", + observedRevision: observed.revisions[0] ?? "", + previousRevision, +phaseControlled: false + } + }; + } finally { + controller.abort(); + await reader.cancel().catch(() => undefined); + await reading; + } +} + +/** + * Superseded by `observeStageFiveSample`: the jitter-based observer is gone, and + * with it any claim that random trigger delays sample the poll interval. + */ +function medianOf(values: readonly number[]): number { + const ordered = [...values].sort((left, right) => left - right); + const middle = Math.floor(ordered.length / 2); + return ordered.length % 2 === 1 ? ordered[middle]! : (ordered[middle - 1]! + ordered[middle]!) / 2; +} + +interface StageSixSeven { + /** Stage 6: browser notification -> the APPLICATION accepted a coherent model. */ + notificationToCoherentModelMs: number; + /** Stage 7: coherent model accepted -> its Home content rendered + presentation. */ + coherentModelToUsefulRenderMs: number | null; + acceptedRevision: string; + acceptedSequence: number; + /** The browser notification that caused the fetch cycle delivering the model. */ + notifiedRevision: string; + latestNotifiedRevision: string; + /** Which paired API fetch delivered the accepted revision ("snapshot"/"runtime-map"). */ + fetchDeliveredBy: string; + fetchStartedAt: number; + /** Null for acceptance-only cells (no Home repaint is declared for them). */ + renderCommitMs: number | null; + presentationFrameMs: number | null; + metricLabel: string; + beforeMetricValue: string | null; + afterMetricValue: string | null; + expectedMetricValue: string | null; + metricChanged: boolean | null; + triggerAt?: number; + triggerAcceptedSequence?: number; + triggerRevision?: string; +} + +/** + * Arm stages 6/7 BEFORE the publication change is triggered. Arming records the + * pre-change Home metric value (so "the DOM changed" is measured rather than + * assumed) and lets the probe wait for the NEXT accepted revision, identified by + * its monotonic sequence number rather than by "any revision other than the last + * one", so a publication that lands between arming and the trigger cannot be + * mistaken for the sample's own. + */ +async function armStageSixSeven( + page: any, + input: { mode: "content" | "acceptance-only"; expectedMetricValue: string; awaitPublicationTrigger?: boolean } +): Promise { + const previousSeq = await page.evaluate("window.__dockermapBenchHelpers.currentAcceptedSeq()"); + await page.evaluate( + `window.__benchInput = ${JSON.stringify({ + mode: input.mode, + previousSeq, + limit: 60_000, + metricLabel: HomeMetricLabel, + expectedMetricValue: input.expectedMetricValue, + awaitPublicationTrigger: Boolean(input.awaitPublicationTrigger) + })}` + ); + await page.evaluate("window.__dockermapBenchHelpers.armModelAcceptance(window.__benchInput)"); + await page.waitForFunction("window.__dockermapBenchHelpers.armed()", undefined, { timeout: 10_000 }); +} + +async function markStageSixSevenPublicationTriggered(page: any): Promise> { + return (await page.evaluate("window.__dockermapBenchHelpers.markModelPublicationTriggered()")) as Record; +} + +async function expectStageSixSevenModelRevision(page: any, revision: string): Promise { + await page.evaluate(`window.__benchInput = ${JSON.stringify({ revision })}`); + await page.evaluate("window.__dockermapBenchHelpers.setExpectedModelRevision(window.__benchInput.revision)"); +} + +/** + * Observe the exact publication a control trigger is meant to cause. This is a + * bounded observation, not a delay: fixture, daemon and API must all expose the + * expected inventory before the sample is accepted as control evidence. + */ +async function observeControlPublication(input: { + fixtureSocket: string; + daemonPort: number; + apiPort: number; + expectedExited: number; + generation: number; + timeoutMs?: number; +}): Promise> { + const startedAt = nowMs(); + const deadline = Date.now() + (input.timeoutMs ?? 60_000); + let fixtureExited = -1; + let daemonExited = -1; + let apiExited = -1; + let daemonRevision = ""; + let apiRevision = ""; + let daemonRuntimeRevision = ""; + let apiRuntimeRevision = ""; + while (Date.now() < deadline) { + const fixture = await getUnix(input.fixtureSocket, "/containers/json"); + const daemon = await fetchJson(`http://127.0.0.1:${input.daemonPort}/daemon/snapshot`, 3_000); + const daemonRuntime = await fetchJson(`http://127.0.0.1:${input.daemonPort}/daemon/runtime/map`, 3_000); + const api = await fetchJson(`http://127.0.0.1:${input.apiPort}/api/snapshot`, 3_000); + const apiRuntime = await fetchJson(`http://127.0.0.1:${input.apiPort}/api/runtime/map`, 3_000); + fixtureExited = countExited(fixture); + daemonExited = countExited(JSON.stringify(daemon)); + apiExited = countExited(JSON.stringify(api)); + daemonRevision = String(daemon?.modelRevision ?? ""); + apiRevision = String(api?.modelRevision ?? ""); + daemonRuntimeRevision = String(daemonRuntime?.modelRevision ?? ""); + apiRuntimeRevision = String(apiRuntime?.modelRevision ?? ""); + if ( + fixtureExited === input.expectedExited && + daemonExited === input.expectedExited && + apiExited === input.expectedExited && + daemonRevision.length > 0 && + daemonRevision === daemonRuntimeRevision && + apiRevision.length > 0 && + apiRevision === apiRuntimeRevision + ) { + return { + generation: input.generation, + expectedExited: input.expectedExited, + fixtureExited, + daemonExited, + apiExited, + daemonRevision, + apiRevision, + daemonRuntimeRevision, + apiRuntimeRevision, + observedAtMs: nowMs(), + elapsedMs: nowMs() - startedAt + }; + } + await sleep(25); + } + throw new Error( + `control ${input.generation} publication did not converge to ${input.expectedExited} exited within ${input.timeoutMs ?? 60_000} ms ` + + `(fixture=${fixtureExited}, daemon=${daemonExited}@${daemonRevision || "none"}/${daemonRuntimeRevision || "none"}, ` + + `api=${apiExited}@${apiRevision || "none"}/${apiRuntimeRevision || "none"})` + ); +} + +async function awaitModelAcceptance(page: any): Promise { + const measured = (await page.evaluate( + "window.__dockermapBenchHelpers.awaitModelAcceptance()" + )) as StageSixSeven | undefined; + if (!measured) throw new Error("model acceptance probe returned no measurement"); + return measured; +} + +async function measureCommandQuery(page: any, preferredToken: string, timeoutMs = 20_000): Promise { + await page.evaluate(`window.__benchInput = ${JSON.stringify({ limit: timeoutMs, preferredToken })}`); + const measured = await page.evaluate( + "window.__dockermapBenchHelpers.commandQuery(window.__benchInput.limit, window.__benchInput.preferredToken)" + ); + if (!measured) throw new Error("command query probe returned no measurement"); + return measured as number; +} + +async function measureProductionBundle(browser: any, webOrigin: string): Promise { + // Cold context: no cache, fresh navigation, production build. +const context = await browser.newContext({ viewport: { width: 1440, height: 900 } }); +recordLifecycle("context_create", "production_bundle"); + try { + const page = await context.newPage(); + recordLifecycle("page_create", "production_bundle"); + await page.goto(`${webOrigin}/`, { waitUntil: "load" }); + recordLifecycle("navigate", "production_bundle"); + // The cold context deliberately has no instrumentation: this stage measures + // the production load itself, so it reads the Navigation Timing entry only. + const duration = await page.evaluate( + "(() => { const entry = performance.getEntriesByType('navigation')[0]; return entry ? entry.duration : 0; })()" + ); + if (!duration) throw new Error("could not read the production navigation duration"); + return duration; + } finally { + await context.close(); + recordLifecycle("closure", "production_bundle_context_and_page"); + } +} + +async function main(): Promise { + // Calibration is a separate retained diagnostic. It never gates ordinary + // Baseline-4 capture or chooses a per-metric conditioning count. +const startedAt = Date.now(); + // A full artifact is forbidden until both independent controls have cleared: + // lifecycle longevity and the Stage-5 exact-publication phase mechanism. + if (!calibration) run("npm", ["run", "perf:preconditioning"]); + if (!calibration) run("npm", ["run", "perf:phase-control"]); +// One private API port for the whole capture: the production web build bakes + // its API origin at build time. It is reserved from the OS, not fixed. + const apiPort = await reservePort(); + process.stdout.write(`[capture] preflight builds (api origin http://127.0.0.1:${apiPort})\n`); + run("npm", ["run", "build", "--workspace", "@dockermap/contracts"]); + run("npm", ["run", "build", "--workspace", "@dockermap/web"], { + VITE_API_BASE_URL: `http://127.0.0.1:${apiPort}` + }); + run("npx", ["vite", "build", "--config", "tests/perf/benchVite.config.mjs"]); + // Benchmark-MODE application build: the same real app with the acceptance seam + // compiled in (see tests/perf/benchAppVite.config.mjs). It is served only to the + // page that measures coherent-model acceptance and its independence control. + run("npx", ["vite", "build", "--config", "tests/perf/benchAppVite.config.mjs"], { + VITE_API_BASE_URL: `http://127.0.0.1:${apiPort}` + }); + assertBuildIsolation(); + + const workRoot = mkdtempSync(join(tmpdir(), "dockermap-bench-")); + + try { +await withFreshBrowserRuns({ +runs, +launch: async () => { + try { + const browser = await chromium.launch({ args: launchArgs }); + recordLifecycle("browser_launch", "fresh_browser_run"); + return browser; + } catch (error) { + recordLifecycle("exception", "browser_launch", error); + throw error; + } +}, +lifecycle: (event: "browser_launch" | "context_create" | "page_create" | "navigate" | "teardown_start" | "teardown_end" | "exception" | "closure", step?: string, error?: unknown) => recordLifecycle(event, step, error), +run: async (browser: any, runIndex: number) => { + for (const plan of plans) { + process.stdout.write( + `[capture] run ${runIndex + 1}/${runs} fixture ${plan.name} (${plan.containers} containers)\n` + ); + // Only the daemon port is chosen here; every other listener either takes an + // atomic OS-assigned port (static servers) or retries on a fresh one. + let daemonPort = 0; + const workdir = join(workRoot, `${plan.name}-${runIndex}`); + mkdirSync(workdir, { recursive: true }); + const projectRoot = join(workdir, "compose-project"); + writeComposeProject(projectRoot, plan.scenario); + // Scenario premise, asserted rather than named: the slow-but-bounded Compose + // fixture must actually present the declared project. Without this an empty or + // truncated tree would still record a cell and look like a fast projection. + if (plan.scenario === "slow-bounded-compose-projection") { + const declaredProject = readFileSync(join(projectRoot, "compose.yaml"), "utf8"); + const serviceCount = declaredProject + .split("\n") + .filter((line) => /^ {2}[A-Za-z0-9._-]+:$/.test(line)).length; + if (serviceCount !== SLOW_COMPOSE_SERVICES) { + throw new Error( + `the slow-Compose premise failed: the project declares ${serviceCount} services, expected ${SLOW_COMPOSE_SERVICES}` + ); + } + } + const benchSink = join(workdir, "bench.jsonl"); + const fixtureSocket = join(workdir, "fixture.sock"); + const fixtureReady = join(workdir, "fixture.ready"); + + let fixtureChild: any = null; + let daemonChild: any = null; + let apiChild: any = null; + let webServer: any = null; + let probeServer: any = null; + let benchAppServer: any = null; +let publicationTracker: PublicationTracker | null = null; + let publicationController: Awaited> | null = null; + let context: any = null; + let benchContext: any = null; + let probeContext: any = null; + try { + fixtureChild = spawnOwned(process.execPath, [ + "tests/perf/fake-docker-api.mjs", + "--socket", + fixtureSocket, + "--containers", + String(plan.containers), + "--scenario", + plan.scenario, + "--project-root", + projectRoot, + "--ready-file", + fixtureReady + ]); + const fixtureDeadline = Date.now() + 15_000; + while (!existsSync(fixtureReady) && Date.now() < fixtureDeadline) await sleep(50); + if (!existsSync(fixtureReady)) throw new Error("fixture Docker daemon did not become ready"); + + // Stage 1: process start -> listener ready. + const emptyPath = join(workdir, "empty-path"); + if (plan.name === "unavailable-optional-provider") mkdirSync(emptyPath, { recursive: true }); + const daemonEnv = { + DOCKERMAP_DOCKER_GATEWAY_SOCKET: fixtureSocket, + DOCKERMAP_BENCH_STAGE_TIMING_PATH: benchSink, + DOCKERMAP_DAEMON_HOST: "127.0.0.1", + DOCKERMAP_PROJECT_ROOT: projectRoot, + ...(plan.name === "unavailable-optional-provider" ? { PATH: emptyPath } : {}) + }; + const healthUrl = (port: number) => `http://127.0.0.1:${port}/daemon/health`; + const daemonReady = async (port: number) => Boolean(await fetchJson(healthUrl(port), 1_000)); + const startedDaemon = await startChildOnFreePort({ + name: "daemon", + spawnOn: (port) => spawnOwned(daemonBinary, [], { ...daemonEnv, DOCKERMAP_DAEMON_PORT: String(port) }), + ping: daemonReady + }); + daemonChild = startedDaemon.child; +daemonPort = startedDaemon.port; + // Test-fixture-only barrier: the production API keeps its ordinary 2000 ms + // poller and sees the daemon through this loopback proxy only during capture. + publicationController = await startStageFivePublicationController({ upstream: `http://127.0.0.1:${daemonPort}` }); + // Stage 5's phase control needs the daemon's publication grid, and the + // tracker needs several publications to fit it. Start it here, with the + // daemon, so the grid is known long before the browser stages begin. + if (hasStage(plan.name, "notificationToCoherentModelMs")) { + publicationTracker = await startPublicationTracker( + daemonPort, + ((await fetchJson(healthUrl(daemonPort), 5_000))?.modelRevision as string | undefined) ?? "", + pollIntervalMs + ); + } + + // Stages 1 and 2 need a CLEAN start per warmed sample, so they are + // measured by restarting the daemon `samples` times on private ports + // rather than by reusing the resident benchmark daemon. + const needsStartup = hasStage(plan.name, "daemonStartToListenerMs"); + if (needsStartup) { + const starts: number[] = []; + const models: number[] = []; + for (let index = 0; index < samples; index += 1) { + let startAt = 0; + const probe = await startChildOnFreePort({ + name: "daemon startup probe", + spawnOn: (port) => { + // The start instant is taken at the successful spawn: a retried + // attempt never contributes a sample. + startAt = nowMs(); + return spawnOwned(daemonBinary, [], { + ...daemonEnv, + // These transient cold-start probes must never write into the + // warmed attribution sink: their first observation is a + // cold-start sample, and mixing it into a stage documented as + // "warmed" would be a provenance defect. + DOCKERMAP_BENCH_STAGE_TIMING_PATH: join(workdir, "probe-bench.jsonl"), + DOCKERMAP_DAEMON_PORT: String(port) + }); + }, + ping: daemonReady + }); + try { + starts.push(nowMs() - startAt); + const readyAt = nowMs(); + await waitForJson( + healthUrl(probe.port), + (value) => value.mode === "docker" && Boolean(value.modelRevision), + 60_000 + ); + models.push(nowMs() - readyAt); + } finally { + stopOwned(probe.child); + } + } + record(plan.name, "daemonStartToListenerMs", starts); + record(plan.name, "listenerToFirstDockerModelMs", models); + } + + // Stages 3, 4, 9: bench attribution from the current implementation. + const needsBench = BENCH_STAGE_KEYS.some((key) => hasStage(plan.name, key)); + if (needsBench) { + const required = calibration ? samples : samples + TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS; + const benchSamples = await waitForBenchSamples(benchSink, required, 300_000); + for (const key of BENCH_STAGE_KEYS) { + if (!hasStage(plan.name, key)) continue; + if (TIME_TO_ANSWER_STAGE_KIND[key] !== "warmed-repeated") { + record(plan.name, key, benchSamples[key].slice(0, samples)); + continue; + } + const burnInCount = calibration ? 0 : TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS; + const requiredForMetric = samples + burnInCount; + // The fixed burn-in is a deterministic conditioning workload, never selected + // from values. The complete ordered window is retained for audit. + const { warmUps: burnIns, recorded } = splitWarmedObservations(benchSamples[key], samples, burnInCount); + const label = `${plan.name}|${key}|run${runIndex}`; + burnInObservations[label] = burnIns; + warmedObservationWindows[label] = benchSamples[key].slice(0, requiredForMetric); + if (calibration) { record(plan.name, key, recorded); continue; } + record(plan.name, key, recorded); + } + } + + const needsApp = + hasStage(plan.name, "publicationToNodeObservationMs") || + hasStage(plan.name, "notificationToCoherentModelMs") || + hasStage(plan.name, "commandQueryMs") || + hasStage(plan.name, "productionBundleMs"); + if (!needsApp) continue; + // Stage 6/7 observe the application's coherent-model acceptance, which + // only exists in the benchmark-MODE build of the real app. Every other + // browser stage is measured on the ordinary production build. + const needsStageSix = hasStage(plan.name, "notificationToCoherentModelMs"); + webServer = await startStaticServer({ directory: join(REPO_ROOT, "apps/web/dist"), port: 0 }); + probeServer = await startStaticServer({ + directory: join(REPO_ROOT, "tests/perf/.bench-dist"), + port: 0 + }); + const webOrigin = webServer.url; + if (needsStageSix) { + benchAppServer = await startStaticServer({ + directory: join(REPO_ROOT, "tests/perf/.bench-app-dist"), + port: 0 + }); + } + // The API port is baked into the production build, so it must stay fixed + // for the whole capture; the retry therefore re-spawns on the SAME port + // (bounded) instead of moving to a new one. + const apiHealth = async () => + Boolean(await fetchJson(`http://127.0.0.1:${apiPort}/api/health`, 1_000)); + const startedApi = await startChildOnFreePort({ + name: "api", + fixedPort: apiPort, + spawnOn: () => + spawnOwned(process.execPath, [join(REPO_ROOT, "node_modules/tsx/dist/cli.mjs"), "apps/api/src/index.ts"], { + PORT: String(apiPort), + DOCKERMAP_DAEMON_URL: publicationController.url, + // The API must accept both browser origins: the production build for + // Cmd-K and the cold production load, the benchmark-mode build for + // coherent-model acceptance. + DOCKERMAP_ALLOWED_ORIGINS: [webOrigin, benchAppServer?.url].filter(Boolean).join(","), + // Pinned explicitly so the recorded interval and the interval that + // actually ran cannot diverge; this is the API's own default. + DOCKERMAP_SSE_INTERVAL_MS: String(pollIntervalMs) + }), + ping: apiHealth + }); + apiChild = startedApi.child; + + // Browser stages. Every browser stage needs `samples` warmed + // observations per controlled run, so revision-driven stages loop over + // real published revision changes instead of being measured once. +context = await browser.newContext({ viewport: { width: 1440, height: 900 } }); +recordLifecycle("context_create", "production_app"); +const page = await context.newPage(); +recordLifecycle("page_create", "production_app"); + if (process.env.DOCKERMAP_BENCH_DEBUG === "1") { + page.on("console", (message: any) => process.stdout.write(`[browser:${message.type()}] ${message.text()}\n`)); + page.on("requestfailed", (failed: any) => + process.stdout.write(`[browser:requestfailed] ${failed.url()} ${failed.failure()?.errorText ?? ""}\n`) + ); + } + await page.addInitScript({ path: join(REPO_ROOT, "tests/perf/browserProbe.js") }); +await page.goto(`${webOrigin}/`, { waitUntil: "domcontentloaded" }); +recordLifecycle("navigate", "production_app"); + await page.waitForFunction("window.__dockermapBenchHelpers.homeReady()", undefined, { + timeout: 90_000 + }); + + // The benchmark-mode application page, used only for stages 6 and 7. + let benchPage: any = null; + if (needsStageSix) { +benchContext = await browser.newContext({ viewport: { width: 1440, height: 900 } }); +recordLifecycle("context_create", "benchmark_app"); +benchPage = await benchContext.newPage(); +recordLifecycle("page_create", "benchmark_app"); + if (process.env.DOCKERMAP_BENCH_DEBUG === "1") { + benchPage.on("console", (message: any) => + process.stdout.write(`[bench:${message.type()}] ${message.text()}\n`) + ); + benchPage.on("requestfailed", (failed: any) => + process.stdout.write(`[bench:requestfailed] ${failed.url()} ${failed.failure()?.errorText ?? ""}\n`) + ); + } + await benchPage.addInitScript({ path: join(REPO_ROOT, "tests/perf/browserProbe.js") }); +await benchPage.goto(`${benchAppServer.url}/`, { waitUntil: "domcontentloaded" }); +recordLifecycle("navigate", "benchmark_app"); + await benchPage.waitForFunction("window.__dockermapBenchHelpers.homeReady()", undefined, { + timeout: 90_000 + }); + } + + const observationSamples: number[] = []; + // Scenario premises must be asserted: a fixture that silently stops + // exercising its premise would still record samples and pass the gate. + const dockerIds = (snapshot: any) => + (snapshot?.containers ?? []) + .map((container: any) => container.id ?? container.name) + .sort() + .join(","); + const premiseDockerIds = + plan.name === "provider-only-revision-change" + ? dockerIds(await fetchJson(`http://127.0.0.1:${daemonPort}/daemon/snapshot`, 30_000)) + : ""; + const premiseProviderStates = + plan.name === "unavailable-optional-provider" + ? JSON.stringify(await fetchJson(`http://127.0.0.1:${daemonPort}/daemon/runtime/map`, 30_000) ?? {}) + : ""; + const coherentSamples: number[] = []; + const usefulSamples: number[] = []; + const querySamples: number[] = []; + const bundleSamples: number[] = []; + const needsRevisionLoop = needsStageSix; + if (needsRevisionLoop) { + const tracker = publicationTracker; + if (!tracker) { + throw new Error(`${plan.name} declares Stage-6 timing but its revision tracker was never started`); + } + for (let index = 0; index < (calibration ? samples : samples + TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS); index += 1) { + // Generation `g` stops the fixture's first `g` containers, so the + // expected Home metric for the publication this sample triggers is + // derived from the fixture rather than assumed. + const generation = index + 1; + const expectedMetricValue = String(expectedExitedCount(plan.containers, plan.scenario, generation)); + // Calibration retains conditioning observations; it does not assert or publish + // the baseline phase sweep. It uses the real free-running API path so the + // collector is independent of the baseline's controlled-capture validity gate. + // End-to-end Stage-6/7 timing follows the normal real polling path. Stage-5 + // phase control is owned by captureStageFive.ts and is never calibrated or + // emitted from this general section. + const phaseControlled = false; + const { sample, revisions, tickCarriedNewerRevision } = await observeStageFiveSample({ + tracker, + daemonPort, + apiPort, + webOrigin, + intervalMs: pollIntervalMs, + runIndex, + sampleIndex: index, +mode: phaseControlled ? "phase-controlled" : "free-running", + controllerUrl: phaseControlled ? publicationController?.url : undefined, + // Runs after the observation stream is connected and before the + // publication: arming the browser here means the acceptance this + // sample measures is caused by THIS publication, and the generation + // trigger is guaranteed to be inside it. + onConnected: async () => { + if (needsStageSix) { + await armStageSixSeven(benchPage, { + // Provider-only fixtures publish a revision with no inventory + // change: stage 6 ends at acceptance (no Home repaint exists to + // wait for) and stage 7 is not declared for them. +mode: hasStage(plan.name, "coherentModelToUsefulRenderMs") ? "content" : "acceptance-only", +expectedMetricValue: hasStage(plan.name, "coherentModelToUsefulRenderMs") ? expectedMetricValue : "" + }); + } +if (plan.name === "docker-topology-change" || plan.name.startsWith("reference-")) { + // A real published inventory change: the fixture daemon serves a + // new generation, so the daemon must publish a new revision. + const fixtureState = await setFixtureGeneration(fixtureSocket, generation); + appendDiagnostic("fixture-identity.jsonl", { + fixture: plan.name, + run: runIndex, + generation, + fixtureRevision: FIXTURE_REVISION, + fixtureState, + expectedExited: Number(expectedMetricValue), + monotonicTimestamp: nowMs() + }); + return { triggeredAtMs: nowMs() }; + } + // `provider-only-revision-change` and `unavailable-optional-provider` + // need no fixture trigger: their revision advance comes from provider state + // alone, which is exactly what those fixtures characterise. + return { triggeredAtMs: nowMs() }; + } + }); + stageFiveSweep.push({ ...sample, fixture: plan.name, tickCarriedNewerRevision }); + observationSamples.push(sample.observedLatencyMs); + // Fail fast on a control failure: the phase the harness drove did not + // produce the latency the design predicted, so this sample is not a + // measurement of the declared phase. Free-running cells make no such + // claim, so the check does not apply to them. + if (phaseControlled && Math.abs(sample.phaseErrorMs) > POLL_PHASE_CONTROL_TOLERANCE_MS) { + throw new Error( + `stage 5 declared phase ${sample.declaredPhaseMs.toFixed(1)} ms produced ` + + `${sample.observedLatencyMs.toFixed(1)} ms instead of the intended ` + + `${sample.intendedLatencyMs.toFixed(1)} ms (error ${sample.phaseErrorMs.toFixed(1)} ms): ` + + "the publication phase was not controlled " + + `[connected at ${sample.connectedAtMs.toFixed(1)}, predicted publication ` + + `${sample.predictedPublicationAtMs.toFixed(1)} (${(sample.predictedPublicationAtMs - sample.connectedAtMs).toFixed(1)} ms after connect), ` + + `observed publication ${sample.observedPublicationAtMs.toFixed(1)} ` + + `(${(sample.observedPublicationAtMs - sample.predictedPublicationAtMs).toFixed(1)} ms from the prediction), ` + + `observed poll tick ${sample.observedObservationAtMs.toFixed(1)}]` + ); + } + if (needsStageSix) { +const measured = await awaitModelAcceptance(benchPage); + await drainLayerDiagnostics(benchPage, plan.name); + coherentSamples.push(measured.notificationToCoherentModelMs); + if (typeof measured.coherentModelToUsefulRenderMs === "number") { + usefulSamples.push(measured.coherentModelToUsefulRenderMs); + } + // Cross-layer attribution. Stage 5 observes revisions as the API's + // own poller announced them on this connection; the app accepts the + // revision its paired fetches returned, and the daemon is read per + // request — so the accepted revision need not appear in this + // connection's stream while the host is churning (each SSE + // connection also polls on its own phase). The binding provenance + // for stage 6 is the browser-side one the probe records: an API + // fetch delivered that revision to the app, and a browser + // notification preceded that fetch cycle. The overlap is therefore + // recorded as evidence, not enforced as a gate. + const acceptedInApiStream = revisions.includes(measured.acceptedRevision); + if (!acceptedInApiStream) { + process.stdout.write( + `[capture] note: accepted revision ${measured.acceptedRevision} was not on this harness stream's own phase ` + + `(api: ${revisions.join(", ") || "none"}; browser fetched it via ${measured.fetchDeliveredBy})\n` + ); + } + } + } + await tracker.stop(); + if (benchPage) { + // No shortcut: the application must reach the daemon only through the + // API, and the notifications feeding stages 6/7 must come from the + // API's real SSE endpoint — not from a harness-injected channel. + const origins: string[] = await benchPage.evaluate( + "window.__dockermapBenchHelpers.requestOrigins()" + ); + const stream: string = await benchPage.evaluate("window.__dockermapBenchHelpers.streamUrl()"); + const daemonOrigin = `http://127.0.0.1:${daemonPort}`; + const apiOrigin = `http://127.0.0.1:${apiPort}`; + if (origins.includes(daemonOrigin)) { + throw new Error( + `the benchmark-mode page fetched the daemon directly (${daemonOrigin}): the measured path ` + + "is not the real API polling path" + ); + } + if (!origins.includes(apiOrigin) || !stream.startsWith(`${apiOrigin}/api/events/stream`)) { + throw new Error( + `the page's notification path is not the API's real stream (origins: ${origins.join(", ") || "none"}; ` + + `stream: ${stream || "none"})` + ); + } + } + } + if (hasStage(plan.name, "commandQueryMs")) { + for (let index = 0; index < (calibration ? samples : samples + TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS); index += 1) { + // A fresh page per sample: the measurement must be a real closed + // palette opening for the first time, not a still-open dialog. + await page.goto(`${webOrigin}/`, { waitUntil: "domcontentloaded" }); + await page.waitForFunction("window.__dockermapBenchHelpers.homeReady()", undefined, { + timeout: 90_000 + }); + querySamples.push(await measureCommandQuery(page, FixtureTopologyQueryToken)); + } + } + if (hasStage(plan.name, "productionBundleMs")) { + for (let index = 0; index < (calibration ? samples : samples + TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS); index += 1) { + bundleSamples.push(await measureProductionBundle(browser, webOrigin)); + } + } + // Assert the scenario premise actually held for this run. + if (plan.name === "provider-only-revision-change") { + const after = dockerIds(await fetchJson(`http://127.0.0.1:${daemonPort}/daemon/snapshot`, 30_000)); + if (after !== premiseDockerIds) { + throw new Error( + "provider-only-revision-change measured a Docker-driven revision: the fixture inventory changed during the run" + ); + } + } + if (plan.name === "unavailable-optional-provider") { + if (!/"state":"(unavailable|disabled|error|stale|collecting)"/.test(premiseProviderStates)) { + throw new Error( + "unavailable-optional-provider measured a fully-provided model: no optional provider was non-fresh" + ); + } + } + if (coherentSamples.length > 0) { + recordWarmed(plan.name, "notificationToCoherentModelMs", coherentSamples, runIndex); + recordWarmed(plan.name, "coherentModelToUsefulRenderMs", usefulSamples, runIndex); +} +if (querySamples.length > 0) recordWarmed(plan.name, "commandQueryMs", querySamples, runIndex); +if (bundleSamples.length > 0) recordWarmed(plan.name, "productionBundleMs", bundleSamples, runIndex); + + // Stages 8, 10: the real production modules measured in real Chromium + // through the benchmark-only entry. + if ( + hasStage(plan.name, "buildModelMs") || + hasStage(plan.name, "legacyTopologyLayoutMs") + ) { + const snapshot = await fetchJson(`http://127.0.0.1:${apiPort}/api/snapshot`, 30_000); + const runtimeMap = await fetchJson(`http://127.0.0.1:${apiPort}/api/runtime/map`, 30_000); + if (!snapshot || !runtimeMap) throw new Error("could not read the fixture model from the API"); +probeContext = await browser.newContext(); +recordLifecycle("context_create", "module_probe"); +const probePage = await probeContext.newPage(); +recordLifecycle("page_create", "module_probe"); +await probePage.goto(`${probeServer.url}/index.html`, { waitUntil: "domcontentloaded" }); +recordLifecycle("navigate", "module_probe"); + await probePage.waitForFunction("Boolean(window.__dockermapProbe)", undefined, { + timeout: 30_000 + }); + await probePage.evaluate( + `window.__benchInput = ${JSON.stringify({ snapshot, runtimeMap, samples, calibration })}` + ); + const measured = (await probePage.evaluate( + "window.__dockermapProbe.measureModel(window.__benchInput.snapshot, window.__benchInput.runtimeMap, window.__benchInput.samples, window.__benchInput.calibration)" + )) as ProbeMeasurement | undefined; + if (!measured) throw new Error("module probe returned no measurement"); + recordWarmed(plan.name, "buildModelMs", measured.buildModelMs, runIndex); + recordWarmed(plan.name, "legacyTopologyLayoutMs", measured.legacyTopologyLayoutMs, runIndex); + } +} finally { +recordLifecycle("teardown_start", "fixture_run"); + const cleanup = async (label: string, close: (() => Promise) | undefined) => { + if (!close) return; + try { await close(); recordLifecycle("closure", label); } catch (error) { recordLifecycle("exception", `${label}_close`, error); } + }; + await cleanup("module_probe_context_and_page", probeContext ? () => probeContext.close() : undefined); + await cleanup("benchmark_app_context_and_page", benchContext ? () => benchContext.close() : undefined); + await cleanup("production_app_context_and_page", context ? () => context.close() : undefined); + const trackerToClose = publicationTracker; +await cleanup("publication_tracker", trackerToClose ? () => trackerToClose.stop() : undefined); + await cleanup("stage_five_publication_controller", publicationController ? () => publicationController!.close() : undefined); +stopOwned(apiChild); + stopOwned(daemonChild); + stopOwned(fixtureChild); + if (webServer) await webServer.close(); + if (probeServer) await probeServer.close(); +if (benchAppServer) await benchAppServer.close(); +recordLifecycle("teardown_end", "fixture_run"); + } + } + } + }); +} catch (error) { +recordLifecycle("exception", "capture", error); +preserveRaw(String(error)); + throw error; + } finally { + rmSync(workRoot, { recursive: true, force: true }); + } + +/* --- Harness evidence --------------------------------------------------- + * Two things the closed evidence schema deliberately does not carry, written + * beside the artifact so a reviewer can audit them without trusting a summary: + * + * 1. the stage-6/7 independence control (verdict + per-run sample sets + + * per-sample acceptance audit); + * 2. warm-up retention: for every warmed cell, the COMPLETE observation window + * in the order the daemon produced it, with the declared warm-up observations + * at the front and the recorded samples proven equal to the artifact's stored + * run, plus the declared-stationarity ratio of each window. A hidden slow + * warm-up value cannot survive this; + * 3. the stage-5 poll-phase sweep: every sample's declared and observed phase, + * the observed phase curve, and the validity verdict of the declared design. + * + * A seam that cannot demonstrate independence, a warmed window that does not + * match the artifact, or a phase sweep that does not cover the interval + * invalidates the capture: no baseline is produced. + */ + // After the run, re-hash the SAME executable. The daemon is spawned repeatedly + // during a long capture (the resident daemon plus every cold-start probe), so a + // substitution or rebuild mid-run would otherwise be invisible. + assertDaemonBinaryProvenance({ + expectedSha256: environment.daemonBinarySha256, + observedSha256: currentDaemonSha256(), + phase: "after capture" + }); + const daemonBinarySha256After = currentDaemonSha256(); + const daemonBinaryEvidence = { + beforeCapture: daemonBinarySha256Before, + afterCapture: daemonBinarySha256After, + pinnedSha256: environment.daemonBinarySha256, + build: environment.daemonBinaryBuild, + cargoRevision: environment.cargoRevision, + matches: daemonBinarySha256Before === daemonBinarySha256After + }; + process.stdout.write( + `[capture] daemon binary verified before and after the capture: ${daemonBinarySha256After.slice(0, 16)}… ` + + `(matches: ${daemonBinaryEvidence.matches})\n` + ); + if (calibration) { + const cells = TIME_TO_ANSWER_WARM_UP_METRICS.flatMap((metric) => + TIME_TO_ANSWER_CALIBRATION_REFERENCE_FIXTURES.map((fixture) => ({ + fixture, + metric, + observations: raw[fixture]?.[metric]?.[0] ?? [] + })) + ); + const derivationReport = deriveWarmUpCalibrationReport(cells); + const artifact = { + kind: "dockermap-v1/time-to-answer-warm-up-calibration-1", + methodologyVersion: TIME_TO_ANSWER_METHODOLOGY, + authority: "REJECTED_NON_AUTHORITATIVE_FOR_BASELINE_4", + constants: { observations: TIME_TO_ANSWER_CALIBRATION_OBSERVATIONS, measured: TIME_TO_ANSWER_WARMED_SAMPLES, safetyMargin: 2, band: [TIME_TO_ANSWER_STATIONARITY_MIN_RATIO, TIME_TO_ANSWER_STATIONARITY_MAX_RATIO] }, + environment, + daemonBinary: daemonBinaryEvidence, + cells, + derivationReport + }; + const serialized = `${JSON.stringify(artifact, null, 2)}\n`; + writeFileSync(calibrationOutputPath!, serialized); + const derivationReportSha256 = createHash("sha256").update(serialized).digest("hex"); + writeFileSync(`${calibrationOutputPath!}.sha256`, `${derivationReportSha256}\n`); + process.stdout.write(`[calibration] wrote complete derivation report to ${calibrationOutputPath} (sha256 ${derivationReportSha256})\n`); + // Calibration is retained historical diagnostic evidence only. A conflict is + // reported, never promoted into Baseline-4 authority and never blocks capture. + return; + } + const harnessEvidencePath = `${outputPath}.harness-evidence.json`; + const warmUpRetention = TIME_TO_ANSWER_END_TO_END_MATRIX.flatMap(({ fixture, stage }) => { + const runs = raw[fixture]?.[stage]; + if (!runs || runs.length === 0) return []; + const kind = TIME_TO_ANSWER_STAGE_KIND[stage] ?? "warmed-repeated"; + const burnInPerWindow = kind === "warmed-repeated" ? TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS : 0; + return runs.map((recorded, run) => { + const label = `${fixture}|${stage}|run${run}`; + const window = warmedObservationWindows[label] ?? null; + return { + fixture, + stage, + run, + kind, + recordedSampleCount: recorded.length, + observationWindow: window, + observationCount: window ? window.length : recorded.length, + declaredBurnInCount: burnInPerWindow, + burnInIndexRange: window ? [0, burnInPerWindow - 1] : null, + burnIns: burnInObservations[label] ?? null, + burnInsMatchWindow: window + ? JSON.stringify(window.slice(0, burnInPerWindow)) === JSON.stringify(burnInObservations[label] ?? null) +: true, +recordedMatchesArtifact: window + ? JSON.stringify(window.slice(burnInPerWindow, burnInPerWindow + recorded.length)) === + JSON.stringify(recorded) + : true + }; + }); + }); + // Stage-5 poll-phase evidence: every sample's declared and observed phase, the + // observed phase curve, and the per-run arrays the shared math consumes. + const stageFiveByFixture = new Map>(); + for (const sample of stageFiveSweep) { + stageFiveByFixture.set(sample.fixture, [...(stageFiveByFixture.get(sample.fixture) ?? []), sample]); + } + const stageFiveEvidence: Record = {}; + for (const [fixture, samplesForFixture] of stageFiveByFixture) { + const runs = [...new Set(samplesForFixture.map((sample) => sample.runIndex))] + .sort((left, right) => left - right) + .map((runIndex) => + samplesForFixture + .filter((sample) => sample.runIndex === runIndex) + .sort((left, right) => left.sampleIndex - right.sampleIndex) + .map((sample) => sample.observedLatencyMs) + ); + stageFiveEvidence[fixture] = { + changes: POLL_PHASE_DIVISIONS, + samples: samplesForFixture, + phaseMediansMs: phaseMediansMs(runs, Number(environment.ssePollIntervalMs)), + validity: stageFiveValidity[fixture] ?? null + }; + } +const harnessEvidence: { + burnInObservations: Record; + warmUpRetention: typeof warmUpRetention; + stageFive: Record; + daemonBinary: typeof daemonBinaryEvidence; +} = { + burnInObservations, + warmUpRetention, + stageFive: stageFiveEvidence, + daemonBinary: daemonBinaryEvidence + }; + const writeHarnessEvidence = (): void => { + writeFileSync(harnessEvidencePath, `${JSON.stringify(harnessEvidence, null, 2)}\n`); + }; + try { + // Fixed burn-in retention is audited structurally for every warmed end-to-end + // stage: exactly observations 1-60 are retained conditioning work and samples + // 61-75 are the complete published measurement window. + for (const entry of warmUpRetention) { + if (entry.recordedSampleCount !== samples) { + throw new Error( + `stage ${entry.fixture}|${entry.stage} run ${entry.run} recorded ${entry.recordedSampleCount} samples, expected ${samples}` + ); + } + if (entry.kind !== "warmed-repeated") continue; + if (entry.observationWindow === null) { + throw new Error(`no observation window was retained for ${entry.fixture}|${entry.stage} run ${entry.run}`); + } + if (entry.observationCount !== samples + entry.declaredBurnInCount) { + throw new Error( + `warmed stage ${entry.fixture}|${entry.stage} run ${entry.run} kept ${entry.observationCount} observations, ` + + `expected ${samples + entry.declaredBurnInCount}` + ); + } + if (entry.declaredBurnInCount !== TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS) throw new Error(`warmed stage ${entry.fixture}|${entry.stage} run ${entry.run} does not use the fixed 60-observation burn-in`); + if (!entry.burnInsMatchWindow) { + throw new Error( + `warmed stage ${entry.fixture}|${entry.stage} run ${entry.run}: the retained burn-in is not the window's first 60 observations` + ); + } + if (!entry.recordedMatchesArtifact) { + throw new Error( + `warmed stage ${entry.fixture}|${entry.stage} run ${entry.run}: the recorded samples are not observations 61-75` + ); + } + } + // Stage-5 poll-phase design: every declared phase represented, the publication + // phase actually driven to its declared offset, the observation arriving through + // the real API poller path, and an observed spread that spans the interval. A + // sweep confined to a narrow band fails here. + const stageFiveRequired = TIME_TO_ANSWER_REFERENCE_FIXTURES.filter((fixture) => + hasStage(fixture.name, "publicationToNodeObservationMs") + ) + .map((fixture) => fixture.name) + .filter((name) => !onlyFixtures || onlyFixtures.includes(name)); + const stageFiveMissing = stageFiveRequired.filter((name) => !stageFiveByFixture.has(name)); + if (stageFiveMissing.length > 0) { + throw new Error(`the stage-5 poll-phase sweep did not run for ${stageFiveMissing.join(", ")}`); + } + for (const fixture of stageFiveRequired) { + const samplesForFixture = stageFiveByFixture.get(fixture)!; + const phaseControlled = isPhaseControlledFixture(fixture); + if (!phaseControlled) { + // Free-running cell: the harness cannot place these publications, so the + // coverage and direction guards do not apply — only that the samples are real + // poller observations, with the phase they achieved recorded. + const freeRunning = assertFreeRunningPhaseSamples( + samplesForFixture, + runs >= TIME_TO_ANSWER_CONTROLLED_RUNS ? 30 : 10 + ); + stageFiveValidity[fixture] = { phaseControlled: false, ...freeRunning }; + stageFiveEvidence[fixture] = { + ...(stageFiveEvidence[fixture] as Record), + phaseControlled: false, + validity: stageFiveValidity[fixture] + }; + process.stdout.write( + `[capture] stage 5 free-running ${fixture}: ${freeRunning.samples} samples, observed latency ` + + `${freeRunning.minObservedLatencyMs.toFixed(1)}–${freeRunning.maxObservedLatencyMs.toFixed(1)} ms ` + + `(no declared phase: these publications are host-provider driven)\n` + ); + continue; + } + const verdict = assertPollPhaseSweep( + samplesForFixture, + Number(environment.ssePollIntervalMs), + // Debug runs may declare a single controlled run, which cannot reach the + // two-samples-per-phase the full protocol requires. The declared minimum is + // therefore relaxed ONLY for probing runs, which can never emit an artifact + // (the closed matrix requires three runs); every other guard still applies. + runs >= TIME_TO_ANSWER_CONTROLLED_RUNS ? POLL_PHASE_MIN_SAMPLES_PER_PHASE : 1 + ); + stageFiveValidity[fixture] = { phaseControlled: true, ...verdict }; + stageFiveEvidence[fixture] = { + ...(stageFiveEvidence[fixture] as Record), + phaseControlled: true, + validity: stageFiveValidity[fixture] + }; + process.stdout.write( + `[capture] stage 5 sweep ${fixture}: ${verdict.samples} samples over ${verdict.divisions} declared phases, ` + + `span ${verdict.spanMs.toFixed(1)} ms (${(verdict.spanShare * 100).toFixed(1)} % of the interval), ` + + `worst phase error ${verdict.worstPhaseErrorMs.toFixed(1)} ms, phase medians ` + + `${verdict.fastestPhaseMedianMs.toFixed(1)}–${verdict.slowestPhaseMedianMs.toFixed(1)} ms\n` + ); + } +} catch (error) { +writeHarnessEvidence(); +preserveRaw(String(error)); + throw error; + } + writeHarnessEvidence(); + process.stdout.write(`[capture] harness evidence at ${harnessEvidencePath}\n`); + + try { + const records = TIME_TO_ANSWER_END_TO_END_MATRIX.map(({ fixture, stage }) => ({ +fixture, +stage, +runs: raw[fixture]?.[stage] ?? [] +})); + if (baselinePath) throw new Error("promotion requires the assembled composite evidence, not the incomplete end-to-end section"); + writeFileSync(outputPath, JSON.stringify({ environment, records }, null, 2)); + process.stdout.write( + `[capture] wrote ${outputPath} in ${((Date.now() - startedAt) / 60_000).toFixed(1)} min (fixture revision ${FIXTURE_REVISION})\n` + ); + } catch (error) { + // Assembly or validation failed: the measurement pass is expensive, so the + // raw samples are preserved even though no artifact can be emitted. + preserveRaw(String(error)); + throw error; + } + } + +await main(); diff --git a/tests/perf/captureIndependence.ts b/tests/perf/captureIndependence.ts new file mode 100644 index 00000000..6dd53822 --- /dev/null +++ b/tests/perf/captureIndependence.ts @@ -0,0 +1,105 @@ +#!/usr/bin/env node +/** + * Controlled Stage-6/7 independence protocol. This is deliberately a complete + * harness rather than a mode of capture.ts: controls must never become + * Baseline-4 observations. + */ +import { appendFileSync, existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, rmSync, writeFileSync } from "node:fs"; +import { spawn, spawnSync } from "node:child_process"; +import { createHash } from "node:crypto"; +import { request } from "node:http"; +import { tmpdir } from "node:os"; +import { join, resolve } from "node:path"; +import { chromium } from "playwright"; +import { + TIME_TO_ANSWER_CONTROLLED_RUNS, TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS, + TIME_TO_ANSWER_INDEPENDENCE_SAMPLES, TIME_TO_ANSWER_METHODOLOGY, + TIME_TO_ANSWER_WARMED_SAMPLES, assertDaemonBinaryProvenance, + assertStageSixSevenIndependence, assertTimeToAnswerEnvironment +} from "../../apps/web/src/lib/performance/timeToAnswerEvidence"; +import { reservePort, startStaticServer } from "./staticServer.mjs"; + +const ROOT = resolve(new URL("../..", import.meta.url).pathname); +const fixtures = [{ name: "reference-25", containers: 25 }, { name: "reference-100", containers: 100 }, { name: "reference-250", containers: 250 }, { name: "docker-topology-change", containers: 25, scenario: "docker-topology-change" }]; +const sleep = (ms: number) => new Promise((done) => setTimeout(done, ms)); +const now = () => Number(process.hrtime.bigint()) / 1e6; +function values(argv: string[]) { return Object.fromEntries(argv.flatMap((value, index, all) => value.startsWith("--") ? [[value.slice(2), all[index + 1] ?? ""]] : [])); } +function git(...args: string[]) { const result = spawnSync("git", args, { cwd: ROOT, encoding: "utf8" }); if (result.status) throw new Error(result.stderr); return result.stdout.trim(); } +function raw(directory: string, value: unknown) { const target = join(directory, "controlled-stage6-stage7-seam-isolation"); mkdirSync(target, { recursive: true }); appendFileSync(join(target, "evidence.jsonl"), `${JSON.stringify(value)}\n`); } +function run(command: string, args: string[], env: NodeJS.ProcessEnv = {}) { const result = spawnSync(command, args, { cwd: ROOT, env: { ...process.env, ...env }, encoding: "utf8", maxBuffer: 64 * 1024 * 1024 }); if (result.status !== 0) throw new Error(`${command} ${args.join(" ")} failed: ${result.stderr || result.stdout}`); } +function spawnOwned(command: string, args: string[], env: NodeJS.ProcessEnv = {}) { const child = spawn(command, args, { cwd: ROOT, env: { ...process.env, ...env }, detached: true, stdio: "ignore" }); return child; } +function stop(child: any) { if (!child?.pid || child.exitCode !== null || child.signalCode) return; try { process.kill(-child.pid, "SIGKILL"); } catch { try { process.kill(child.pid, "SIGKILL"); } catch {} } } +async function json(url: string, timeout = 2_000): Promise { const controller = new AbortController(); const timer = setTimeout(() => controller.abort(), timeout); try { const response = await fetch(url, { signal: controller.signal }); return response.ok ? response.json() : null; } catch { return null; } finally { clearTimeout(timer); } } +async function wait(url: string, predicate: (body: any) => boolean, timeout = 60_000): Promise { const deadline = Date.now() + timeout; while (Date.now() < deadline) { const body = await json(url); if (body && predicate(body)) return body; await sleep(25); } throw new Error(`timed out waiting for ${url}`); } +function postUnix(socketPath: string, path: string) { return new Promise((done, fail) => { const call = request({ socketPath, path, method: "POST" }, (response) => response.statusCode === 200 ? done() : fail(new Error(`fixture trigger failed: ${response.statusCode}`))); call.on("error", fail); call.end(); }); } +function contains(directory: string, needle: string): boolean { for (const entry of readdirSync(directory, { withFileTypes: true })) { const file = join(directory, entry.name); if (entry.isDirectory() ? contains(file, needle) : /\.(js|mjs|html)$/.test(entry.name) && readFileSync(file, "utf8").includes(needle)) return true; } return false; } +function assertIsolation() { if (contains(join(ROOT, "apps/web/dist"), "__dockermapBenchAcceptanceSink")) throw new Error("production build contains benchmark acceptance seam"); if (!contains(join(ROOT, "tests/perf/.bench-app-dist"), "__dockermapBenchAcceptanceSink")) throw new Error("benchmark build lacks real acceptance seam"); } + +type Sample = { stageSixMs: number; stageSevenMs: number; acceptedRevision: string; snapshotRevision: string; runtimeMapRevision: string; acceptanceSeam: "observed"; delayAppliedAfterAcceptance: boolean; delayStartedAfterAcceptance: boolean }; +/** A full isolated fixture/API/browser lifecycle for one controlled sample. */ +export async function runControlledStageSixSeven(input: { fixture: string; containers: number; scenario?: string; control: boolean; delayMs: number; rawDir: string; generation: number; daemonBinary: string; apiPort: number; pollIntervalMs: number; browserFlags: string[] }): Promise { + if ((!input.control && input.delayMs !== 0) || (input.control && input.delayMs !== TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS)) throw new Error("independence delay contract violated"); + if (!Number.isInteger(input.generation) || input.generation < 1 || input.generation > input.containers) throw new Error("fixture generation must remain within the observable container count"); + const work = mkdtempSync(join(tmpdir(), "dockermap-independence-")); + const socket = join(work, "fixture.sock"), ready = join(work, "fixture.ready"); +let fixture: any, daemon: any, api: any, server: any, browser: any, context: any; + const lifecycle: string[] = []; + try { + fixture = spawnOwned(process.execPath, ["tests/perf/fake-docker-api.mjs", "--socket", socket, "--containers", String(input.containers), "--scenario", input.scenario ?? "reference", "--ready-file", ready]); lifecycle.push("fixture"); + const fixtureDeadline = Date.now() + 15_000; while (!existsSync(ready) && Date.now() < fixtureDeadline) await sleep(25); if (!existsSync(ready)) throw new Error("fixture did not become ready"); + const daemonPort = await reservePort(); + daemon = spawnOwned(input.daemonBinary, [], { DOCKERMAP_DOCKER_GATEWAY_SOCKET: socket, DOCKERMAP_DAEMON_HOST: "127.0.0.1", DOCKERMAP_DAEMON_PORT: String(daemonPort) }); lifecycle.push("daemon"); +await wait(`http://127.0.0.1:${daemonPort}/daemon/health`, (body) => Boolean(body.modelRevision)); + server = await startStaticServer({ directory: join(ROOT, "tests/perf/.bench-app-dist"), port: 0 }); lifecycle.push("static"); +api = spawnOwned(process.execPath, [join(ROOT, "node_modules/tsx/dist/cli.mjs"), "apps/api/src/index.ts"], { PORT: String(input.apiPort), DOCKERMAP_DAEMON_URL: `http://127.0.0.1:${daemonPort}`, DOCKERMAP_ALLOWED_ORIGINS: server.url, DOCKERMAP_SSE_INTERVAL_MS: String(input.pollIntervalMs) }); lifecycle.push("api"); + await wait(`http://127.0.0.1:${input.apiPort}/api/health`, Boolean); + browser = await chromium.launch({ args: input.browserFlags }); lifecycle.push("browser"); context = await browser.newContext({ viewport: { width: 1440, height: 900 } }); + const page = await context.newPage(); await page.addInitScript({ path: join(ROOT, "tests/perf/browserProbe.js") }); await page.goto(`${server.url}/`, { waitUntil: "domcontentloaded" }); await page.waitForFunction("window.__dockermapBenchHelpers.homeReady()", undefined, { timeout: 90_000 }); + const previousSeq = await page.evaluate("window.__dockermapBenchHelpers.currentAcceptedSeq()"); +await page.evaluate(`window.__dockermapBenchHelpers.armModelAcceptance(${JSON.stringify({ mode: "content", seamIsolation: true, previousSeq, limit: 60_000, metricLabel: "Offline" })})`); +// This arms a one-shot delay before the fixture changes. The benchmark build +// consumes it only after the real useSystemModel acceptance seam records the +// next coherent pair; no trigger or daemon publication identity is supplied. +if (input.control) await page.evaluate(`window.__dockermapBenchHelpers.armSeamIsolationDelay(${JSON.stringify(input.delayMs)})`); +await postUnix(socket, `/__fixture/topology-generation/${input.generation}`); +const measured: any = await page.evaluate("window.__dockermapBenchHelpers.awaitModelAcceptance()"); +if (!measured.acceptedRevision || measured.acceptedRevision !== measured.snapshotRevision || measured.acceptedRevision !== measured.runtimeMapRevision) throw new Error("accepted snapshot/runtime-map pair was not internally coherent"); +if (input.control && (!Number.isFinite(measured.delayStartedAt) || measured.delayStartedAt < measured.acceptanceAt)) throw new Error("control delay did not begin after the acceptance timestamp"); + const origins: string[] = await page.evaluate("window.__dockermapBenchHelpers.requestOrigins()"); const stream: string = await page.evaluate("window.__dockermapBenchHelpers.streamUrl()"); + if (origins.includes(`http://127.0.0.1:${daemonPort}`) || !origins.includes(`http://127.0.0.1:${input.apiPort}`) || !stream.startsWith(`http://127.0.0.1:${input.apiPort}/api/events/stream`)) throw new Error("benchmark bypassed the real API SSE path"); +const result = { stageSixMs: measured.notificationToCoherentModelMs, stageSevenMs: measured.coherentModelToUsefulRenderMs, acceptedRevision: measured.acceptedRevision, snapshotRevision: measured.snapshotRevision, runtimeMapRevision: measured.runtimeMapRevision, acceptanceSeam: "observed" as const, delayAppliedAfterAcceptance: input.delayMs === 0 || measured.coherentModelToUsefulRenderMs >= input.delayMs, delayStartedAfterAcceptance: input.delayMs === 0 || measured.delayStartedAt >= measured.acceptanceAt }; + if (!Number.isFinite(result.stageSixMs) || !Number.isFinite(result.stageSevenMs) || !result.delayAppliedAfterAcceptance) throw new Error("invalid acceptance or post-acceptance delay proof"); +raw(input.rawDir, { fixture: input.fixture, control: input.control, generation: input.generation, lifecycle, measured, result, at: now() }); + return result; + } catch (error) { raw(input.rawDir, { fixture: input.fixture, control: input.control, generation: input.generation, lifecycle, verdict: "FAIL", error: String(error) }); throw error; } +finally { try { await context?.close(); } finally { try { await browser?.close(); } finally { try { await server?.close(); } finally { stop(api); stop(daemon); stop(fixture); rmSync(work, { recursive: true, force: true }); } } } } +} + +export async function main() { + const input = values(process.argv.slice(2)); + if (!input.metadata || !input.output || !input["raw-dir"] || !input.checkpoint) throw new Error("--metadata, --output, --raw-dir and --checkpoint are required"); + const metadata = JSON.parse(readFileSync(input.metadata, "utf8")); assertTimeToAnswerEnvironment(metadata.environment); + if (git("status", "--porcelain")) throw new Error("refusing independence protocol from a dirty worktree"); + if (git("rev-parse", "HEAD") !== input.checkpoint || metadata.environment.sourceRevision !== input.checkpoint || metadata.environment.harnessRevision !== git("log", "-1", "--format=%H", "--", "tests/perf", "apps/web/src/lib/performance")) throw new Error("checkpoint must exactly bind metadata, harness and HEAD"); + if (metadata.environment.methodologyVersion !== TIME_TO_ANSWER_METHODOLOGY) throw new Error("metadata methodology mismatch"); + const daemonBinary = metadata.daemonBinary ?? join(ROOT, "crates/target/release/dockermap-daemon"); const digest = () => createHash("sha256").update(readFileSync(daemonBinary)).digest("hex"); + assertDaemonBinaryProvenance({ expectedSha256: metadata.environment.daemonBinarySha256, observedSha256: digest(), phase: "before independence capture" }); + const apiPort = await reservePort(); + run("npm", ["run", "build", "--workspace", "@dockermap/contracts"]); run("npm", ["run", "build", "--workspace", "@dockermap/web"]); run("npx", ["vite", "build", "--config", "tests/perf/benchAppVite.config.mjs"], { VITE_API_BASE_URL: `http://127.0.0.1:${apiPort}` }); assertIsolation(); + const cells: any[] = []; + try { + for (const fixture of fixtures) { + const normal: Sample[] = [], control: Sample[] = []; + // Every sample owns a fresh fixture lifecycle, so generation one is a + // discriminating transition on every fixture and cannot saturate its + // Offline count during the fixed 48-sample protocol. + for (let runIndex = 0; runIndex < TIME_TO_ANSWER_CONTROLLED_RUNS; runIndex++) for (let sample = 0; sample < TIME_TO_ANSWER_WARMED_SAMPLES; sample++) normal.push(await runControlledStageSixSeven({ fixture: fixture.name, containers: fixture.containers, scenario: fixture.scenario, control: false, delayMs: 0, rawDir: input["raw-dir"], generation: 1, daemonBinary, apiPort, pollIntervalMs: Number(metadata.environment.ssePollIntervalMs), browserFlags: metadata.environment.browserFlags })); + for (let sample = 0; sample < TIME_TO_ANSWER_INDEPENDENCE_SAMPLES; sample++) control.push(await runControlledStageSixSeven({ fixture: fixture.name, containers: fixture.containers, scenario: fixture.scenario, control: true, delayMs: TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS, rawDir: input["raw-dir"], generation: 1, daemonBinary, apiPort, pollIntervalMs: Number(metadata.environment.ssePollIntervalMs), browserFlags: metadata.environment.browserFlags })); + const verdict = assertStageSixSevenIndependence({ fixture: fixture.name, normalStageSixMs: normal.map((s) => s.stageSixMs), normalStageSevenMs: normal.map((s) => s.stageSevenMs), controlStageSixMs: control.map((s) => s.stageSixMs), controlStageSevenMs: control.map((s) => s.stageSevenMs) }); +cells.push({ fixture: fixture.name, normalStageSixRuns: Array.from({ length: TIME_TO_ANSWER_CONTROLLED_RUNS }, (_, runIndex) => normal.slice(runIndex * TIME_TO_ANSWER_WARMED_SAMPLES, (runIndex + 1) * TIME_TO_ANSWER_WARMED_SAMPLES).map((s) => s.stageSixMs)), normalStageSevenRuns: Array.from({ length: TIME_TO_ANSWER_CONTROLLED_RUNS }, (_, runIndex) => normal.slice(runIndex * TIME_TO_ANSWER_WARMED_SAMPLES, (runIndex + 1) * TIME_TO_ANSWER_WARMED_SAMPLES).map((s) => s.stageSevenMs)), controlEvidence: { delayMs: TIME_TO_ANSWER_INDEPENDENCE_DELAY_MS, acceptanceSeam: "observed", samples: control, verdict } }); + } + assertDaemonBinaryProvenance({ expectedSha256: metadata.environment.daemonBinarySha256, observedSha256: digest(), phase: "after independence capture" }); +writeFileSync(input.output, `${JSON.stringify({ methodologyVersion: TIME_TO_ANSWER_METHODOLOGY, checkpointSha: input.checkpoint, measurementProtocol: "controlled-stage6-stage7-seam-isolation", cells, independenceEvidence: { verdict: "PASS", controlSamplesExcludedFromBaselineTiming: true, limitation: "publication-level causal identity unavailable", validatesDaemonToBrowserAttribution: false } }, null, 2)}\n`); + } catch (error) { raw(input["raw-dir"], { verdict: "FAIL", error: String(error) }); throw error; } +} +if (import.meta.url === new URL(process.argv[1]!, "file:").href) main().catch((error) => { process.stderr.write(`[independence] FAIL: ${String(error)}\n`); process.exitCode = 1; }); diff --git a/tests/perf/captureStageFive.ts b/tests/perf/captureStageFive.ts new file mode 100644 index 00000000..1f059063 --- /dev/null +++ b/tests/perf/captureStageFive.ts @@ -0,0 +1,79 @@ +#!/usr/bin/env node +/** + * Dedicated controlled Stage-5 sub-benchmark (#335). + * + * This intentionally has no import from capture.ts. It owns the only + * arm -> mark -> trigger -> identity-ack protocol used to make a Stage-5 + * cell, and emits its own raw evidence section for composite assembly. + */ +import { createServer } from "node:http"; +import { writeFileSync } from "node:fs"; +import { + TIME_TO_ANSWER_CONTROLLED_RUNS, + TIME_TO_ANSWER_METHODOLOGY, + TIME_TO_ANSWER_CONTROLLED_POLL_MATRIX, + TIME_TO_ANSWER_WARMED_SAMPLES +} from "../../apps/web/src/lib/performance/timeToAnswerEvidence"; +import { POLL_PHASE_CONTROL_TOLERANCE_MS, declaredPhaseForSample } from "../../apps/web/src/lib/performance/timeToAnswerPollPhase"; +import { armStageFivePublication, startStageFivePublicationController } from "./stageFivePublicationControl.mjs"; + +const intervalMs = 2_000; +const sleep = (ms: number) => new Promise((done) => setTimeout(done, ms)); +const stageFiveFixtures = TIME_TO_ANSWER_CONTROLLED_POLL_MATRIX.map((cell) => cell.fixture); + +function argumentsByName(argv: string[]): Record { + return Object.fromEntries(argv.flatMap((value, index, all) => value.startsWith("--") ? [[value.slice(2), all[index + 1] ?? ""]] : [])); +} + +async function fakeDaemon() { + let revision = "initial"; + const server = createServer((_request, response) => response.end(JSON.stringify({ modelRevision: revision }))); + await new Promise((done) => server.listen(0, "127.0.0.1", done)); + return { + url: `http://127.0.0.1:${(server.address() as any).port}`, + set: (next: string) => { revision = next; }, + close: () => new Promise((done) => server.close(() => done())) + }; +} + +async function measureFixture(fixture: string) { + const daemon = await fakeDaemon(); + const controller = await startStageFivePublicationController({ upstream: daemon.url }); + const samples: Array> = []; + try { + for (let run = 0; run < TIME_TO_ANSWER_CONTROLLED_RUNS; run += 1) { + for (let sample = 0; sample < TIME_TO_ANSWER_WARMED_SAMPLES; sample += 1) { + const previousRevision = run === 0 && sample === 0 ? "initial" : sample === 0 ? `${fixture}-${run - 1}-${TIME_TO_ANSWER_WARMED_SAMPLES - 1}` : `${fixture}-${run}-${sample - 1}`; + const triggerId = `${fixture}-${run}-${sample}`; + const requestedPhaseMs = declaredPhaseForSample(run, sample, intervalMs); + await fetch(`${controller.url}/daemon/health`); + const publication = await armStageFivePublication({ controllerUrl: controller.url, triggerId, requestedPhaseMs, previousRevision, timeoutMs: intervalMs * 4 }); + const polls: Array<{ at: number; revision: string }> = []; + const timer = setInterval(async () => { + const at = performance.now(); + const payload = await (await fetch(`${controller.url}/daemon/health`)).json() as { modelRevision: string }; + polls.push({ at, revision: payload.modelRevision }); + }, intervalMs); + const revision = `${fixture}-${run}-${sample}`; + const ack: any = await publication.release(() => daemon.set(revision)); + const deadline = Date.now() + intervalMs * 4; + while (Date.now() < deadline && !polls.some((poll) => poll.revision === ack.revision)) await sleep(2); + clearInterval(timer); + const observed = polls.find((poll) => poll.revision === ack.revision); + if (!observed || ack.triggerId !== triggerId || ack.revision !== revision) throw new Error(`${fixture} ${triggerId} did not observe the acknowledged publication`); + const observedPhaseMs = ack.releasedAtMs - ack.pollAtMs; + if (Math.abs(observedPhaseMs - requestedPhaseMs) > POLL_PHASE_CONTROL_TOLERANCE_MS) throw new Error(`${fixture} ${triggerId} exceeded phase tolerance`); + samples.push({ run, sample, triggerId, revision, requestedPhaseMs, observedPhaseMs, pollAtMs: ack.pollAtMs, releasedAtMs: ack.releasedAtMs, observedAtMs: observed.at, observedLatencyMs: observed.at - ack.releasedAtMs }); + } + } + return samples; + } finally { await controller.close(); await daemon.close(); } +} + +export async function main() { + const args = argumentsByName(process.argv.slice(2)); + if (!args.output || !args.checkpoint) throw new Error("--output and --checkpoint are required for Stage-5 raw evidence"); + const evidence = await Promise.all(stageFiveFixtures.map(async (fixture) => [fixture, await measureFixture(fixture)] as const)); + writeFileSync(args.output, `${JSON.stringify({ methodologyVersion: TIME_TO_ANSWER_METHODOLOGY, checkpointSha: args.checkpoint, measurementProtocol: "controlled-poll-phase", cells: Object.fromEntries(evidence) }, null, 2)}\n`); +} +if (import.meta.url === new URL(process.argv[1]!, "file:").href) main().catch((error) => { process.stderr.write(`[stage-five] FAIL: ${String(error)}\n`); process.exitCode = 1; }); diff --git a/tests/perf/captureStageFivePlumbing.test.mjs b/tests/perf/captureStageFivePlumbing.test.mjs new file mode 100644 index 00000000..bb7860fd --- /dev/null +++ b/tests/perf/captureStageFivePlumbing.test.mjs @@ -0,0 +1,66 @@ +/** + * Full-capture Stage-5 plumbing integration guard (#335). + * + * Unlike the phase-control gate, this exercises capture's own Stage-5 control + * entrypoint against the controller and a real observer request. It proves the + * capture path reaches the same arm -> mark -> trigger -> exact-ack mechanism. + */ +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { createServer } from "node:http"; +import { resolve } from "node:path"; +import test from "node:test"; +import { armCaptureStageFivePublication } from "./stageFiveCaptureControl.mjs"; +import { startStageFivePublicationController } from "./stageFivePublicationControl.mjs"; + +const root = resolve(new URL("../..", import.meta.url).pathname); + +async function upstream() { + let revision = "before"; + const server = createServer((_request, response) => response.end(JSON.stringify({ modelRevision: revision }))); + await new Promise((done) => server.listen(0, "127.0.0.1", done)); + return { + url: `http://127.0.0.1:${server.address().port}`, + set: (next) => { revision = next; }, + close: () => new Promise((done) => server.close(done)) + }; +} + +test("capture Stage-5 control reaches the shared exact-ack mechanism after its observer connects", async () => { + const daemon = await upstream(); + const controller = await startStageFivePublicationController({ upstream: daemon.url }); + try { + // Capture opens the observer after arming and before its trigger/release. + await fetch(`${controller.url}/daemon/health`); + const publication = await armCaptureStageFivePublication({ + controllerUrl: controller.url, + triggerId: "capture-stage-five-1", + requestedPhaseMs: 20, + previousRevision: "before", + timeoutMs: 1_000 + }); + let observedRevision = ""; + const ackPromise = publication.release(() => daemon.set("capture-triggered")); + // This is the controller-facing equivalent of capture's live API-SSE reader: + // it witnesses only the exact revision released by the shared mechanism. + await new Promise((done) => setTimeout(done, 5)); + await fetch(`${controller.url}/daemon/health`); + const ack = await ackPromise; + observedRevision = (await (await fetch(`${controller.url}/daemon/health`)).json()).modelRevision; + assert.equal(ack.triggerId, "capture-stage-five-1"); + assert.equal(ack.revision, "capture-triggered"); + assert.equal(observedRevision, ack.revision); + } finally { + await controller.close(); + await daemon.close(); + } +}); + +test("capture cannot bypass its shared Stage-5 control entrypoint", () => { + const capture = readFileSync(resolve(root, "tests/perf/capture.ts"), "utf8"); + assert.match(capture, /import \{ armCaptureStageFivePublication \} from "\.\/stageFiveCaptureControl\.mjs"/); + assert.match(capture, /await armCaptureStageFivePublication\(/); + assert.doesNotMatch(capture, /import \{ armStageFivePublication/); + assert.doesNotMatch(capture, /function (?:postControl|waitForPublicationAcknowledgement)/); + assert.doesNotMatch(capture, /__stage-five-control\/(?:arm|mark|ack)/); +}); diff --git a/tests/perf/captureSyntax.test.mjs b/tests/perf/captureSyntax.test.mjs new file mode 100644 index 00000000..5f538032 --- /dev/null +++ b/tests/perf/captureSyntax.test.mjs @@ -0,0 +1,20 @@ +import { readFile } from "node:fs/promises"; +import { transform } from "esbuild"; +import test from "node:test"; + +test("capture harness entrypoint transforms", async () => { + const source = await readFile(new URL("./capture.ts", import.meta.url), "utf8"); + try { + await transform(source, { loader: "ts", format: "esm", target: "es2022" }); + } catch (error) { + const diagnostics = error.errors ?? []; + const details = diagnostics + .map(({ text, location }) => + location + ? `capture.ts:${location.line}:${location.column}: ${text}` + : text + ) + .join("\n"); + throw new Error(`capture.ts failed esbuild transform:\n${details || error.message}`); + } +}); diff --git a/tests/perf/dockerFixtureTopology.mjs b/tests/perf/dockerFixtureTopology.mjs new file mode 100644 index 00000000..25a9a599 --- /dev/null +++ b/tests/perf/dockerFixtureTopology.mjs @@ -0,0 +1,258 @@ +/** + * Deterministic Docker fixture topology for the time-to-answer benchmark + * (#335). One generator, shared by the fixture daemon that the benchmark + * collects from and by anything that needs to describe the same host. + * + * Properties this module deliberately guarantees: + * - deterministic: same (containers, scenario) always yields byte-identical JSON + * - secret-free: every value is generated here; nothing is read from the host + * - contract-bounded: sizes stay inside the published DockerMap caps + * - no host contact: it never talks to Docker, the network, or the filesystem + */ +import { createHash } from "node:crypto"; + +export const FIXTURE_REVISION = "dockermap-v1/time-to-answer-fixtures-1"; + +/** Reference container counts from the issue. */ +export const REFERENCE_SIZES = [25, 100, 250]; + +/** The closed fixture set the evidence contract measures. */ +export const FIXTURE_NAMES = [ + "reference-25", + "reference-100", + "reference-250", + "provider-only-revision-change", + "docker-topology-change", + "slow-bounded-compose-projection", + "unavailable-optional-provider" +]; + +const SCENARIOS = [ + "reference", + "provider-only-revision-change", + "docker-topology-change", + "slow-bounded-compose-projection", + "unavailable-optional-provider" +]; + +/** Bounded, deterministic CPU-only cost used by the slow-Compose scenario. */ +export const SLOW_COMPOSE_SERVICES = 400; + +function digest(seed) { + return createHash("sha256").update(seed).digest("hex"); +} + +function id(seed) { + return digest(seed); +} + +function name(index) { + return `dockermap-fixture-${String(index).padStart(4, "0")}`; +} + +function networkName(index) { + return index === 0 ? "dockermap-fixture-internal" : `dockermap-fixture-net-${index}`; +} + +function pad(value) { + return String(value).padStart(2, "0"); +} + +/** + * Build the container summary list for a size and scenario. + * + * `topologyGeneration` lets a scenario publish a changed inventory from the + * SAME fixture daemon without touching anything else, so the only difference the + * daemon observes is the Docker model. Generation `g` stops the first `g` + * containers: a bounded, deterministic, strictly monotone inventory delta whose + * effect is visible in the product (the Home attention/offline metrics change on + * every generation), which is what makes the stage-7 "expected content for the + * newly accepted model" check discriminating rather than vacuous. Generation 0 is + * the pristine, all-running inventory for every fixture. + */ +export function buildContainers( + containers, + scenario = "reference", + topologyGeneration = 0, + projectRoot = "/srv/dockermap-fixture" +) { + if (!Number.isInteger(containers) || containers < 1 || containers > 250) { + throw new Error("Fixture container count must be an integer between 1 and 250."); + } + if (!SCENARIOS.includes(scenario)) throw new Error(`Unknown fixture scenario: ${scenario}`); + if (!Number.isInteger(topologyGeneration) || topologyGeneration < 0) { + throw new Error("Fixture topology generation must be a non-negative integer."); + } + const suffix = topologyGeneration === 0 ? "" : `-g${topologyGeneration}`; + const list = []; + for (let index = 0; index < containers; index += 1) { + const base = `${scenario}${suffix}/${index}`; + const label = (key) => `${key}`; + const networks = { + [networkName(index % 5)]: { + NetworkID: id(`network/${index % 5}`), + EndpointID: id(`endpoint/${base}`), + Gateway: "172.30.0.1", + IPAddress: `172.30.${Math.floor(index / 250) + 1}.${(index % 250) + 2}`, + IPPrefixLen: 24, + MacAddress: `02:42:ac:1e:00:${pad((index % 250) + 2)}`, + Aliases: [name(index)] + } + }; + const ports = []; + const mounts = []; + if (index % 11 === 0) { + ports.push({ + IP: "0.0.0.0", + PrivatePort: 80, + PublicPort: 8000 + (index % 1000), + Type: "tcp" + }); + } + if (index % 17 === 0) { + ports.push({ IP: "127.0.0.1", PrivatePort: 443, PublicPort: 9000 + (index % 1000), Type: "tcp" }); + } + if (index % 13 === 0) { + mounts.push({ + Type: "bind", + Source: "/var/run/docker.sock", + Destination: "/var/run/docker.sock", + Mode: "", + RW: true + }); + } + if (index % 5 === 0) { + mounts.push({ + Type: "volume", + Name: `dockermap-fixture-vol-${index % 25}`, + Source: `/var/lib/docker/volumes/dockermap-fixture-vol-${index % 25}/_data`, + Destination: "/data", + Mode: "", + RW: true + }); + } + const labels = { + "com.dockermap.fixture": label(`${scenario}${suffix}`), + "com.dockermap.fixture.index": String(index) + }; + if (index % 7 === 0) { + // A bounded subset carries a complete Compose identity so the private + // Compose/runtime binding path has deterministic candidates. + labels["com.docker.compose.project"] = "dockermap-fixture"; + labels["com.docker.compose.service"] = `fixture-service-${index % 40}`; + labels["com.docker.compose.config-hash"] = digest(`config-hash/${index % 40}`).slice(0, 64); + labels["com.docker.compose.project.config_files"] = `${projectRoot}/compose.yaml`; + } + const exited = index < topologyGeneration; + list.push({ + Id: id(`container/${base}`), + Names: [`/${name(index)}`], + Image: "dockermap/fixture:1", + ImageID: id("image/1").slice(0, 12), + Command: "/bin/dockermap-fixture", + Created: 1_700_000_000 + index, + Ports: ports, + Labels: labels, + State: exited ? "exited" : "running", + Status: exited ? "Exited (0) 1 hours ago" : "Up 1 hours", + HostConfig: { NetworkMode: "bridge" }, + NetworkSettings: { Networks: networks }, + Mounts: mounts + }); + } + return list; +} + +export function buildNetworks(containers) { + const count = Math.min(5, Math.max(1, Math.ceil(containers / 50))); + const list = []; + for (let index = 0; index < count; index += 1) { + list.push({ + Name: networkName(index), + Id: id(`network/${index}`), + Created: "2026-09-23T00:00:00.000000000Z", + Scope: "local", + Driver: "bridge", + EnableIPv6: false, + IPAM: { + Driver: "default", + Options: null, + Config: [{ Subnet: "172.30.0.0/24", Gateway: "172.30.0.1" }] + }, + Internal: index === 0, + Attachable: false, + Ingress: false, + ConfigFrom: { Network: "" }, + ConfigOnly: false, + Containers: {}, + Options: {}, + Labels: { "com.dockermap.fixture": "network" } + }); + } + return list; +} + +export function buildVolumes(containers) { + const count = Math.max(1, Math.ceil(containers / 5)); + const list = []; + for (let index = 0; index < count; index += 1) { + list.push({ + CreatedAt: "2026-09-23T00:00:00Z", + Driver: "local", + Labels: { "com.dockermap.fixture": "volume" }, + Mountpoint: `/var/lib/docker/volumes/dockermap-fixture-vol-${index}/_data`, + Name: `dockermap-fixture-vol-${index}`, + Options: {}, + Scope: "local" + }); + } + return list; +} + +/** One deterministic Compose project tree for the slow-but-bounded scenario. */ +export function buildSlowComposeProject(services = SLOW_COMPOSE_SERVICES) { + const lines = ["name: dockermap-fixture", "services:"]; + for (let index = 0; index < services; index += 1) { + lines.push(` fixture-service-${index}:`); + lines.push(" image: dockermap/fixture:1"); + lines.push(" volumes:"); + lines.push(` - ./fixture-data-${index}:/data`); + lines.push(" depends_on:"); + lines.push(` - fixture-service-${(index + 1) % services}`); + } + return `${lines.join("\n")}\n`; +} + +export function buildTopology({ + containers, + scenario = "reference", + topologyGeneration = 0, + projectRoot = "/srv/dockermap-fixture" +}) { + return { + revision: FIXTURE_REVISION, + scenario, + containers: buildContainers(containers, scenario, topologyGeneration, projectRoot), + networks: buildNetworks(containers), + volumes: buildVolumes(containers) + }; +} + +/** + * How many containers the fixture reports as exited for a generation, i.e. how + * many services the product must present as offline / needing attention. + * + * Derived from `buildContainers` rather than reimplemented, so the harness + * expectation the stage-7 check asserts against the rendered Home metrics can + * never drift from the inventory the fixture daemon actually served. + */ +export function expectedExitedCount( + containers, + scenario = "reference", + topologyGeneration = 0, + projectRoot = "/srv/dockermap-fixture" +) { + return buildContainers(containers, scenario, topologyGeneration, projectRoot).filter( + (container) => container.State === "exited" + ).length; +} diff --git a/tests/perf/dockerFixtureTopology.test.mjs b/tests/perf/dockerFixtureTopology.test.mjs new file mode 100644 index 00000000..52b9da51 --- /dev/null +++ b/tests/perf/dockerFixtureTopology.test.mjs @@ -0,0 +1,126 @@ +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { + FIXTURE_NAMES, + FIXTURE_REVISION, + REFERENCE_SIZES, + SLOW_COMPOSE_SERVICES, + buildContainers, + buildNetworks, + buildSlowComposeProject, + buildTopology, + buildVolumes, + expectedExitedCount +} from "./dockerFixtureTopology.mjs"; + +describe("deterministic Docker fixture topology", () => { + it("is byte-identical for the same size and scenario", () => { + for (const size of REFERENCE_SIZES) { + assert.equal( + JSON.stringify(buildTopology({ containers: size })), + JSON.stringify(buildTopology({ containers: size })) + ); + } + }); + + it("publishes exactly the requested container count with unique identities", () => { + for (const size of REFERENCE_SIZES) { + const containers = buildContainers(size); + assert.equal(containers.length, size); + assert.equal(new Set(containers.map((container) => container.Id)).size, size); + assert.equal(new Set(containers.map((container) => container.Names[0])).size, size); + } + }); + + it("stays inside the published contract bounds and is secret-free", () => { + const topology = buildTopology({ containers: 250 }); + assert.ok(topology.volumes.length <= 250); + assert.ok(topology.networks.length <= 5); + for (const container of topology.containers) { + assert.match(container.Id, /^[a-f0-9]{64}$/); + assert.ok(container.Mounts.length <= 2); + assert.ok(container.Ports.length <= 2); + for (const value of Object.values(container.Labels)) { + assert.doesNotMatch(String(value), /(password|secret|token|api[-_]?key)/i); + } + } + assert.equal(topology.revision, FIXTURE_REVISION); + }); + + it("rejects unsupported sizes and scenarios instead of inventing a host", () => { + assert.throws(() => buildContainers(0), /between 1 and 250/); + assert.throws(() => buildContainers(251), /between 1 and 250/); + assert.throws(() => buildContainers(25.5), /between 1 and 250/); + assert.throws(() => buildContainers(25, "not-a-scenario"), /Unknown fixture scenario/); + }); + + it("changes only the Docker inventory when the topology generation advances", () => { + const before = buildTopology({ containers: 100, scenario: "docker-topology-change" }); + const after = buildTopology({ + containers: 100, + scenario: "docker-topology-change", + topologyGeneration: 1 + }); + assert.equal(before.containers.length, after.containers.length); + assert.notDeepEqual( + before.containers.map((container) => container.Id), + after.containers.map((container) => container.Id) + ); + // Networks and volumes are unchanged, so the observed change is Docker topology only. + assert.deepEqual(before.networks, after.networks); + assert.deepEqual(before.volumes, after.volumes); + }); + + it("publishes a strictly monotone, product-visible delta per generation", () => { + // Stage 7 asserts the accepted model's expected Home content against the + // rendered metrics, which is only discriminating if each generation actually + // changes what the product renders: generation `g` stops the first `g` + // containers, so the offline/attention count is `g` and always differs from + // the previous generation. + assert.equal(expectedExitedCount(100, "reference", 0), 0); + for (let generation = 1; generation <= 15; generation += 1) { + const count = expectedExitedCount(100, "reference", generation); + assert.equal(count, generation); + assert.notEqual(count, expectedExitedCount(100, "reference", generation - 1)); + } + // Every fixture starts pristine, and the derivation matches the served list. + for (const scenario of ["reference", "docker-topology-change", "unavailable-optional-provider"]) { + assert.equal(expectedExitedCount(25, scenario, 0), 0); + const containers = buildContainers(25, scenario, 4); + assert.equal(expectedExitedCount(25, scenario, 4), containers.filter((entry) => entry.State === "exited").length); + assert.equal(expectedExitedCount(25, scenario, 4), 4); + assert.equal(containers.length, 25); + } + assert.throws(() => buildContainers(25, "reference", -1), /non-negative integer/); + }); + + it("produces a bounded, valid Compose project for the slow-projection scenario", () => { + const project = buildSlowComposeProject(); + assert.match(project, /^name: dockermap-fixture\nservices:\n/); + assert.equal(project.trimEnd().split("\n").length, 2 + SLOW_COMPOSE_SERVICES * 6); + assert.ok(SLOW_COMPOSE_SERVICES <= 400); + }); + + it("declares exactly the fixture names the evidence contract measures", () => { + // Pinned on both sides on purpose: the contract declares these seven names + // and the fixture daemon must be able to serve every one of them. The + // collector validates its artifact against the contract matrix, so any + // drift here fails the capture closed rather than producing a quiet gap. + assert.deepEqual([...FIXTURE_NAMES].sort(), [ + "docker-topology-change", + "provider-only-revision-change", + "reference-100", + "reference-25", + "reference-250", + "slow-bounded-compose-projection", + "unavailable-optional-provider" + ]); + }); + + it("keeps volumes and networks sized from the container count", () => { + assert.equal(buildVolumes(25).length, 5); + assert.equal(buildVolumes(100).length, 20); + assert.equal(buildNetworks(25).length, 1); + assert.equal(buildNetworks(250).length, 5); + }); +}); diff --git a/tests/perf/emit-metadata.mjs b/tests/perf/emit-metadata.mjs new file mode 100644 index 00000000..7ee3553a --- /dev/null +++ b/tests/perf/emit-metadata.mjs @@ -0,0 +1,165 @@ +#!/usr/bin/env node +/** + * Emit the pinned benchmark environment metadata for a controlled capture + * (#335). Every field is read from the runner itself, so a capture cannot be + * recorded against a guessed environment. + * + * node tests/perf/emit-metadata.mjs --output /controlled/time-to-answer-metadata.json + * + * `sourceRevision` is the candidate source revision (git HEAD by default); + * `fixtureRevision` and `ssePollIntervalMs` are the harness constants. + */ +import { execFileSync } from "node:child_process"; +import { createHash } from "node:crypto"; +import { existsSync, readFileSync, writeFileSync } from "node:fs"; +import { createRequire } from "node:module"; +import { resolve } from "node:path"; +import { fileURLToPath } from "node:url"; +import { FIXTURE_REVISION } from "./dockerFixtureTopology.mjs"; + +const REPO_ROOT = resolve(fileURLToPath(new URL("../..", import.meta.url))); +const args = Object.fromEntries( + process.argv.slice(2).flatMap((value, index, all) => (value.startsWith("--") ? [[value.slice(2), all[index + 1] ?? ""]] : [])) +); +const outputPath = args.output; +if (!outputPath || outputPath.startsWith("--")) { + throw new Error("Usage: node tests/perf/emit-metadata.mjs --output "); +} + +/** Reduce a version string to a safe token (the contract's value allowlist). */ +function safeToken(value) { + const token = String(value).trim().replace(/[^A-Za-z0-9._@:+=-]+/g, "-").replace(/^-+|-+$/g, ""); + if (!token) throw new Error("environment field reduced to an empty token"); + return token.slice(0, 160); +} + +function command(file, commandArgs) { + return execFileSync(file, commandArgs, { encoding: "utf8", cwd: REPO_ROOT }).trim(); +} + +const nodeRevision = safeToken(process.version.replace(/^v/, "")); +const rustRevision = safeToken(command("rustc", ["--version"]).split(" ")[1] ?? "unknown"); +let dockerRevision = "unavailable"; +try { + dockerRevision = safeToken(command("docker", ["version", "--format", "{{.Server.Version}}"])); +} catch { + dockerRevision = "unavailable"; +} +const osKernel = safeToken(command("uname", ["-r"])); +const architecture = safeToken(command("uname", ["-m"])); +const cpuClass = safeToken(`cpus-${command("nproc", [])}vcpu`); +const runnerClass = safeToken(args["runner-class"] ?? `linux-${architecture}-dedicated`); +let osImage = safeToken(args["os-image"] ?? "unknown"); +if (!args["os-image"]) { + try { + const lines = readFileSync("/etc/os-release", "utf8").split("\n"); + const name = lines.find((line) => line.startsWith("ID="))?.split("=")[1]?.replace(/"/g, "") ?? "linux"; + const version = lines.find((line) => line.startsWith("VERSION_ID="))?.split("=")[1]?.replace(/"/g, "") ?? ""; + osImage = safeToken(`${name}-${version}`); + } catch { + osImage = "linux-unknown"; + } +} +const fontEnvironment = safeToken(args["font-environment"] ?? "system-default"); + +const require = createRequire(import.meta.url); +const playwrightPackage = JSON.parse( + readFileSync(require.resolve("playwright-core/package.json"), "utf8") +); +const browserRevision = safeToken(playwrightPackage.version); +const browserFlags = ["--disable-background-networking", "--disable-sync", "--no-first-run", "--no-default-browser-check"]; + +let sourceRevision = args["source-revision"]; +if (!sourceRevision) { + try { + sourceRevision = command("git", ["rev-parse", "HEAD"]); + } catch { + sourceRevision = "uncommitted"; + } +} +// The harness revision is the last commit that touched the benchmark itself, so +// a baseline names both the product and the harness that measured it. +let harnessRevision = args["harness-revision"]; +if (!harnessRevision) { + try { + harnessRevision = command("git", ["log", "-1", "--format=%H", "--", "tests/perf", "apps/web/src/lib/performance"]); + } catch { + harnessRevision = "uncommitted"; + } +} +if (!harnessRevision) { + throw new Error("could not resolve the benchmark harness revision; pass --harness-revision"); +} +// Derived from the API source rather than assumed, so the pin cannot silently +// drift from the interval the API actually uses. +const apiSource = readFileSync(resolve(REPO_ROOT, "apps/api/src/index.ts"), "utf8"); +const sseDefault = apiSource.match(/DOCKERMAP_SSE_INTERVAL_MS,\s*([0-9_]+)/); +if (!sseDefault) { + throw new Error("could not derive the API's DOCKERMAP_SSE_INTERVAL_MS default from apps/api/src/index.ts"); +} +const ssePollIntervalMs = safeToken(sseDefault[1].replace(/_/g, "")); +// Build the exact daemon binary this capture will run, then pin its digest. The +// benchmark does not claim bit-for-bit reproducible Rust builds across machines; +// it proves which binary THIS capture executed. The cargo workspace lives under +// crates/, so the manifest path is part of the canonical command. +const DAEMON_BUILD = "cargo build --release --locked -p dockermap-daemon --manifest-path crates/Cargo.toml"; +/** Space-free form for the closed evidence metadata (safe-value constrained). */ +const DAEMON_BUILD_SLUG = "cargo-build-release-locked-p-dockermap-daemon-manifest-path-crates-Cargo-toml"; +/** + * Mirrors `TIME_TO_ANSWER_METHODOLOGY` in + * apps/web/src/lib/performance/timeToAnswerEvidence.ts. This file is plain Node and + * cannot import the TypeScript contract, so the value is duplicated and guarded by + * tests/perf/methodologyDrift.test.mjs, which fails if the two ever diverge. + */ +const METHODOLOGY_VERSION = "dockermap-v1/time-to-answer-methodology-8"; +const daemonBinaryPath = resolve(REPO_ROOT, "crates/target/release/dockermap-daemon"); +try { + command("bash", ["-lc", `cd ${JSON.stringify(REPO_ROOT)} && cargo ${DAEMON_BUILD.replace(/^cargo /, "")}`]); +} catch (error) { + throw new Error(`the release daemon failed to build (${DAEMON_BUILD}): ${error}`); +} +if (!existsSync(daemonBinaryPath)) { + throw new Error(`the release daemon is missing after ${DAEMON_BUILD}`); +} +const daemonBinarySha256 = createHash("sha256").update(readFileSync(daemonBinaryPath)).digest("hex"); +const cargoRevision = safeToken(command("cargo", ["--version"])); +if (!existsSync(resolve(REPO_ROOT, "crates/target/release/dockermap-daemon"))) { + throw new Error( + "the release daemon is missing: run `npm run build:deploy` (or cargo build --release) before capturing" + ); +} + +const metadata = { + environment: { + runnerClass, + cpuClass, + osImage, + osKernel, + nodeRevision, + rustRevision, + dockerRevision, + // Derived from the API's own source default; the capture passes this value + // explicitly to the API so the pin and the running interval cannot diverge. + ssePollIntervalMs, + harnessRevision: safeToken(harnessRevision), + daemonBinarySha256, + daemonBinaryBuild: DAEMON_BUILD_SLUG, + cargoRevision, + browserEngine: "chromium", + browserRevision, + browserFlags, + fontEnvironment, + buildMode: "production", + fixtureRevision: FIXTURE_REVISION, + sourceRevision: safeToken(sourceRevision), + // The measurement design this metadata pins. It must equal + // TIME_TO_ANSWER_METHODOLOGY in the contract (asserted by + // tests/perf/methodologyDrift.test.mjs and enforced again by the capture). + methodologyVersion: METHODOLOGY_VERSION + }, + daemonBinary: resolve(REPO_ROOT, "crates/target/release/dockermap-daemon") +}; + +writeFileSync(outputPath, `${JSON.stringify(metadata, null, 2)}\n`); +process.stdout.write(`${JSON.stringify(metadata.environment, null, 2)}\n`); +process.stdout.write(`[metadata] wrote ${outputPath}\n`); diff --git a/tests/perf/fake-docker-api.mjs b/tests/perf/fake-docker-api.mjs new file mode 100644 index 00000000..e0b75c82 --- /dev/null +++ b/tests/perf/fake-docker-api.mjs @@ -0,0 +1,142 @@ +#!/usr/bin/env node +/** + * Deterministic fake Docker Engine API for the time-to-answer benchmark (#335). + * + * It serves only the three inventory endpoints the daemon's read-only + * collector uses, from the shared fixture topology generator, over a unix + * socket. It never touches a real Docker daemon, the network, or the host + * filesystem beyond its own socket. + * + * Usage: + * node tests/perf/fake-docker-api.mjs --socket /tmp/fake.sock \ + * --containers 100 --scenario reference [--ready-file /tmp/fake.ready] + * + * Control (benchmark harness only, not a Docker route): + * POST /__fixture/topology-generation/ -> bump the published generation + * GET /__fixture/state -> current generation and counts + */ +import { createServer } from "node:http"; +import { unlinkSync, writeFileSync } from "node:fs"; +import { buildContainers, buildNetworks, buildVolumes, FIXTURE_REVISION } from "./dockerFixtureTopology.mjs"; + +const args = Object.fromEntries( + process.argv + .slice(2) + .flatMap((value, index, all) => (value.startsWith("--") ? [[value.slice(2), all[index + 1]]] : [])) +); +const socketPath = args.socket; +if (!socketPath || socketPath.startsWith("--")) { + throw new Error("Usage: fake-docker-api.mjs --socket --containers [--scenario ] [--ready-file ]"); +} +const containers = Number(args.containers ?? 25); +const scenario = args.scenario ?? "reference"; +const projectRoot = args["project-root"] ?? "/srv/dockermap-fixture"; +if (!Number.isInteger(containers) || containers < 1 || containers > 250) { + throw new Error("--containers must be an integer between 1 and 250."); +} + +const state = { generation: 0, requests: 0 }; + +function json(response, status, body) { + const payload = JSON.stringify(body); + response.writeHead(status, { + "content-type": "application/json", + "content-length": Buffer.byteLength(payload) + }); + response.end(payload); +} + +const server = createServer((request, response) => { + state.requests += 1; + const raw = request.url ?? "/"; + const [path, query = ""] = raw.split("?"); + // bollard addresses the versioned path; accept both shapes. + const route = path.replace(/^\/v\d+(?:\.\d+)?/, "") || "/"; + + const control = /^\/__fixture\/topology-generation\/(\d+)$/.exec(path); + if (control && request.method === "POST") { + state.generation = Number(control[1]); + json(response, 200, { generation: state.generation }); + return; + } + if (path === "/__fixture/state") { + json(response, 200, { ...state, containers, scenario, fixtureRevision: FIXTURE_REVISION }); + return; + } + if (route === "/_ping") { + response.writeHead(200, { + "content-type": "text/plain", + "api-version": "1.44", + "docker-experimental": "false", + "ostype": "linux" + }); + response.end("OK"); + return; + } + if (route === "/version") { + json(response, 200, { + Platform: { Name: "DockerMap fixture" }, + Components: [{ Name: "Engine", Version: "29.0.0", Details: { ApiVersion: "1.44", Os: "linux" } }], + Version: "29.0.0", + ApiVersion: "1.44", + MinAPIVersion: "1.24", + GitCommit: "fixture", + GoVersion: "fixture", + Os: "linux", + Arch: "amd64", + KernelVersion: args.kernel ?? "fixture", + BuildTime: "2026-09-23T00:00:00.000000000+00:00" + }); + return; + } + if (route === "/info") { + json(response, 200, { + ID: "FIXTURE:DOCKER:ENGINE", + Containers: containers, + ContainersRunning: containers, + ContainersPaused: 0, + ContainersStopped: 0, + Images: 1, + Driver: "overlay2", + ServerVersion: "29.0.0", + OperatingSystem: "DockerMap fixture", + OSType: "linux", + Architecture: "x86_64", + NCPU: 4, + MemTotal: 8_589_934_592, + Name: "dockermap-fixture" + }); + return; + } + if (route === "/containers/json") { + json(response, 200, buildContainers(containers, scenario, state.generation, projectRoot)); + return; + } + if (route === "/networks") { + json(response, 200, buildNetworks(containers)); + return; + } + if (route === "/volumes") { + json(response, 200, { Volumes: buildVolumes(containers), Warnings: [] }); + return; + } + json(response, 404, { + message: `fixture daemon does not serve ${request.method} ${raw} (query: ${query})` + }); +}); + +try { + unlinkSync(socketPath); +} catch { + // no stale socket +} +server.listen(socketPath, () => { + if (args["ready-file"]) writeFileSync(args["ready-file"], String(process.pid)); + process.stdout.write(`fake-docker-api listening on ${socketPath} (${containers} containers, ${scenario})\n`); +}); + +for (const signal of ["SIGINT", "SIGTERM"]) { + process.on(signal, () => { + server.close(() => process.exit(0)); + }); +} diff --git a/tests/perf/methodologyDrift.test.mjs b/tests/perf/methodologyDrift.test.mjs new file mode 100644 index 00000000..34bc4b7e --- /dev/null +++ b/tests/perf/methodologyDrift.test.mjs @@ -0,0 +1,40 @@ +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { resolve } from "node:path"; +import test from "node:test"; + +const root = resolve(new URL("../..", import.meta.url).pathname); +const read = (path) => readFileSync(resolve(root, path), "utf8"); + +test("metadata and contract pin the same methodology", () => { + const contract = read("apps/web/src/lib/performance/timeToAnswerEvidence.ts"); + const emitter = read("tests/perf/emit-metadata.mjs"); + const version = contract.match(/TIME_TO_ANSWER_METHODOLOGY = "([^"]+)"/)?.[1]; + assert.equal(emitter.match(/METHODOLOGY_VERSION = "([^"]+)"/)?.[1], version); +}); + +test("ordinary end-to-end capture uses fixed 60 plus 15 without calibration authority", () => { + const contract = read("apps/web/src/lib/performance/timeToAnswerEvidence.ts"); + const capture = read("tests/perf/capture.ts"); + assert.match(contract, /TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS = 60/); + assert.doesNotMatch(contract, /TIME_TO_ANSWER_FROZEN_WARM_UP_COUNTS/); + assert.doesNotMatch(capture, /frozenWarmUpCount|assertWarmUpStationarity/); + assert.match(capture, /splitWarmedObservations\(values, samples\)/); + assert.match(capture, /TIME_TO_ANSWER_END_TO_END_BURN_IN_OBSERVATIONS \+ samples/); + assert.match(capture, /burnInObservations\[label\] = burnIns/); +}); + +test("calibration remains persisted historical diagnostic evidence", () => { + const capture = read("tests/perf/capture.ts"); + assert.match(capture, /kind: "dockermap-v1\/time-to-answer-warm-up-calibration-1"/); + assert.match(capture, /Calibration is retained historical diagnostic evidence only/); + assert.doesNotMatch(capture, /derivationReport\.verdict === "CONFLICT"\) throw/); +}); + +test("controlled protocols remain separate from normal end-to-end timing", () => { + const capture = read("tests/perf/capture.ts"); + const assembler = read("tests/perf/assembleCompositeEvidence.ts"); + assert.match(capture, /TIME_TO_ANSWER_END_TO_END_MATRIX/); + assert.match(assembler, /controlled-poll-phase/); +assert.match(assembler, /controlled-stage6-stage7-seam-isolation/); +}); diff --git a/tests/perf/package.json b/tests/perf/package.json new file mode 100644 index 00000000..3dbc1ca5 --- /dev/null +++ b/tests/perf/package.json @@ -0,0 +1,3 @@ +{ + "type": "module" +} diff --git a/tests/perf/phaseControl.ts b/tests/perf/phaseControl.ts new file mode 100644 index 00000000..85dd3144 --- /dev/null +++ b/tests/perf/phaseControl.ts @@ -0,0 +1,71 @@ +#!/usr/bin/env node +/** Runnable Stage-5 publication-control pre-gate (#335). */ +import { createServer } from "node:http"; +import { armStageFivePublication, startStageFivePublicationController } from "./stageFivePublicationControl.mjs"; +import { POLL_PHASE_CONTROL_TOLERANCE_MS, pollPhaseGridMs } from "../../apps/web/src/lib/performance/timeToAnswerPollPhase"; + +const intervalMs = 2_000; +const sleep = (ms: number) => new Promise((done) => setTimeout(done, ms)); + +async function fixture() { + let revision = "initial"; + const server = createServer((_request, response) => response.end(JSON.stringify({ modelRevision: revision }))); + await new Promise((done) => server.listen(0, "127.0.0.1", done)); + return { + url: `http://127.0.0.1:${(server.address() as any).port}`, + set: (next: string) => { revision = next; }, + close: () => new Promise((done) => server.close(() => done())) + }; +} + +async function runFixture(name: string) { + const daemon = await fixture(); + const controller = await startStageFivePublicationController({ upstream: daemon.url }); + const grid = pollPhaseGridMs(intervalMs); + const latencies: number[] = []; + try { + for (const [index, requestedPhaseMs] of grid.entries()) { + const previousRevision = index === 0 ? "initial" : `${name}-trigger-${index - 1}`; + const triggerId = `${name}-trigger-${index}`; +await fetch(`${controller.url}/daemon/health`); +const publication = await armStageFivePublication({ +controllerUrl: controller.url, +triggerId, +requestedPhaseMs, +previousRevision, +timeoutMs: intervalMs * 4 +}); + // This is the real cadence under test: a Node-style fixed setInterval poller, + // not a predicted grid or a random achieved phase distribution. + const polls: Array<{ at: number; revision: string }> = []; + const timer = setInterval(async () => { + const at = performance.now(); + const payload = await (await fetch(`${controller.url}/daemon/health`)).json() as { modelRevision: string }; + polls.push({ at, revision: payload.modelRevision }); + }, intervalMs); +const ack: any = await publication.release(() => { daemon.set(`${name}-trigger-${index}`); }); +const deadline = Date.now() + intervalMs * 4; + while (Date.now() < deadline && !polls.some((poll) => poll.revision === ack?.revision)) await sleep(5); + clearInterval(timer); + const observed = polls.find((poll) => poll.revision === ack?.revision); + if (!ack || !observed) throw new Error(`${name} phase ${requestedPhaseMs} did not observe its exact trigger`); + const observedPhaseMs = ack.releasedAtMs - ack.pollAtMs; + const phaseErrorMs = observedPhaseMs - requestedPhaseMs; + const latency = observed.at - ack.releasedAtMs; + if (ack.triggerId !== triggerId || ack.revision !== `${name}-trigger-${index}`) throw new Error(`${name} observed a substituted publication`); + if (Math.abs(phaseErrorMs) > POLL_PHASE_CONTROL_TOLERANCE_MS) throw new Error(`${name} phase error ${phaseErrorMs} exceeds tolerance`); + latencies.push(latency); + process.stdout.write(`[phase-control] ${name} intended phase ${requestedPhaseMs.toFixed(1)} ms; observed phase ${observedPhaseMs.toFixed(1)} ms; trigger identity ${triggerId}; phase error ${phaseErrorMs.toFixed(1)} ms\n`); + } + const span = Math.max(...latencies) - Math.min(...latencies); + if (span < intervalMs * 0.5) throw new Error(`${name} grid span ${span.toFixed(1)} ms is below 50% of the polling interval`); + return span; + } finally { await controller.close(); await daemon.close(); } +} + +export async function main() { + const fixtures = ["reference-25", "reference-100", "reference-250", "provider-only-revision-change", "docker-topology-change", "unavailable-optional-provider"]; + const spans = await Promise.all(fixtures.map(runFixture)); + process.stdout.write(`[phase-control] PASS: ${fixtures.join(", ")}; grid span ${spans.map((span) => span.toFixed(1)).join(", ")} ms; tolerance ${POLL_PHASE_CONTROL_TOLERANCE_MS} ms\n`); +} +if (import.meta.url === new URL(process.argv[1]!, "file:").href) main().catch((error) => { process.stderr.write(`[phase-control] FAIL: ${String(error)}\n`); process.exitCode = 1; }); diff --git a/tests/perf/preconditioning.ts b/tests/perf/preconditioning.ts new file mode 100644 index 00000000..c465a133 --- /dev/null +++ b/tests/perf/preconditioning.ts @@ -0,0 +1,21 @@ +#!/usr/bin/env node +/** Runnable reduced sustained-preconditioning gate for issue #335. */ +import { runSequentialPreconditioning } from "./preconditioningLifecycle.mjs"; + +export async function main(): Promise { + const startedAt = performance.now(); + const result = await runSequentialPreconditioning(); + const elapsedMs = performance.now() - startedAt; + process.stdout.write( + `[preconditioning] PASS: ${result.runs} sequential fresh-browser runs; ` + + `${result.fixtures.join(", ")}; generations ${result.generations.join(", ")}; ` + + `${elapsedMs.toFixed(0)} ms\n` + ); +} + +if (import.meta.url === new URL(process.argv[1]!, "file:").href) { + main().catch((error: unknown) => { + process.stderr.write(`[preconditioning] FAIL: ${String(error)}\n`); + process.exitCode = 1; + }); +} diff --git a/tests/perf/preconditioningLifecycle.mjs b/tests/perf/preconditioningLifecycle.mjs new file mode 100644 index 00000000..389661e0 --- /dev/null +++ b/tests/perf/preconditioningLifecycle.mjs @@ -0,0 +1,64 @@ +/** + * Reduced, deterministic longevity control for the benchmark harness. It does + * not capture performance evidence; it exercises the ownership boundaries that + * must survive the long controlled matrix. + */ +import { withFreshBrowserRuns } from "./browserLifecycle.mjs"; + +export const PRECONDITIONING_RUNS = 12; +export const PRECONDITIONING_FIXTURES = ["reference-25", "reference-100"]; +export const PRECONDITIONING_GENERATIONS = [16, 17, 18]; + +export function assertBalancedLifecycle(events) { + const created = new Set(); + const tornDown = new Set(); + for (const event of events) { + if (event.event === "create") created.add(event.id); + if (event.event === "teardown") tornDown.add(event.id); + } + if (created.size !== tornDown.size || [...created].some((id) => !tornDown.has(id))) { + throw new Error(`preconditioning lifecycle leak: created ${created.size}, torn down ${tornDown.size}`); + } +} + +export async function runSequentialPreconditioning({ + runs = PRECONDITIONING_RUNS, + launch, + createReader = async () => ({ async cancel() {} }) +} = {}) { + if (!Number.isInteger(runs) || runs < 9 || runs > 16) { + throw new Error(`preconditioning requires 9-16 sequential runs, got ${runs}`); + } + const launchBrowser = launch ?? (async () => { + const { chromium } = await import("playwright"); + return chromium.launch(); + }); + const events = []; + let readerSequence = 0; + await withFreshBrowserRuns({ + runs, + launch: launchBrowser, + lifecycle: (event, _kind, id) => { + if (event === "create") events.push({ event: "create", id }); + if (event === "closure") events.push({ event: "teardown", id }); + }, + run: async (_browser, run) => { + for (const fixture of PRECONDITIONING_FIXTURES) { + for (const generation of PRECONDITIONING_GENERATIONS) { + const id = `reader-${readerSequence++}`; + events.push({ event: "create", id, fixture, run, generation }); + const reader = await createReader({ fixture, run, generation, id }); + try { + // A stage-5 observation has no useful result until its stream is closed. + await Promise.resolve(); + } finally { + await reader.cancel(); + events.push({ event: "teardown", id, fixture, run, generation }); + } + } + } + } + }); + assertBalancedLifecycle(events); + return { runs, fixtures: PRECONDITIONING_FIXTURES, generations: PRECONDITIONING_GENERATIONS, events }; +} diff --git a/tests/perf/probe/entry.ts b/tests/perf/probe/entry.ts new file mode 100644 index 00000000..e76ef748 --- /dev/null +++ b/tests/perf/probe/entry.ts @@ -0,0 +1,60 @@ +/** + * Benchmark-only probe entry (#335). + * + * It imports the REAL production modules — `buildModel` from + * apps/web/src/lib/model.ts and `layoutServices` from apps/web/src/lib/layout.ts + * — and exposes them for measurement in real Chromium. Nothing is copied or + * reimplemented here, and this entry is built by a benchmark-only Vite config: + * the ordinary production build never includes it. + * + * The probe itself performs no timing until the capture harness calls + * `measureModel`, so it cannot benchmark anything by merely being loaded. + */ +import { buildModel } from "../../../apps/web/src/lib/model"; +import { layoutServices } from "../../../apps/web/src/lib/layout"; + +interface ProbeApi { + measureModel( + snapshot: unknown, + runtimeMap: unknown, + samples: number + , calibration?: boolean): Promise<{ buildModelMs: number[]; legacyTopologyLayoutMs: number[] }>; +} + +declare global { + interface Window { + __dockermapProbe?: ProbeApi; + } +} + + function timed(samples: number, run: () => void): number[] { + const measured: number[] = []; + for (let index = 0; index < samples; index += 1) { + const start = performance.now(); + run(); + measured.push(performance.now() - start); + } + return measured; +} + +window.__dockermapProbe = { + async measureModel(snapshot, runtimeMap, samples, calibration = false) { + const build = () => { + buildModel(snapshot as never, runtimeMap as never); + }; + const model = buildModel(snapshot as never, runtimeMap as never); + const layout = () => { + layoutServices(model.services, model.relationships, (service, index) => `${service.id}\u0000${index}`); + }; + // Ordinary capture returns all 75 ordered calls so the harness can retain its + // 60-call burn-in and publish only calls 61-75. Calibration remains a + // separately requested 60-observation diagnostic series. + const count = calibration ? samples : samples + 60; + return { + buildModelMs: timed(count, build), + legacyTopologyLayoutMs: timed(count, layout) + }; + } +}; + +document.getElementById("probe-status")!.textContent = "ready"; diff --git a/tests/perf/probe/index.html b/tests/perf/probe/index.html new file mode 100644 index 00000000..7ddcc82f --- /dev/null +++ b/tests/perf/probe/index.html @@ -0,0 +1,11 @@ + + + + + DockerMap benchmark probe + + +

loading

+ + + diff --git a/tests/perf/productionIsolation.test.mjs b/tests/perf/productionIsolation.test.mjs new file mode 100644 index 00000000..8f8cd075 --- /dev/null +++ b/tests/perf/productionIsolation.test.mjs @@ -0,0 +1,251 @@ +/** + * Production isolation proof for the time-to-answer benchmark (#335). + * + * The benchmark may never leak into the shipped product. These checks fail + * closed: the ordinary production artifact must not contain the acceptance seam, + * the benchmark probe entry or any benchmark identifier; the seam's identifiers + * may appear in product source only inside the seam module and its two import + * sites; the production Vite config must define the compile-time flag `false`; + * and no production build path may reach the benchmark configs. + * + * They require a production build of the web app to exist, because the strongest + * evidence is the shipped bundle itself (`npm run build`, which `npm run check` + * runs before this suite). + */ +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { existsSync, readFileSync, readdirSync, statSync } from "node:fs"; +import { join, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; + +const REPO_ROOT = resolve(fileURLToPath(new URL("../..", import.meta.url))); + +/** + * Harness-only identifiers. They live in `tests/perf` and must never appear in a + * product source file, in the production bundle, or in a benchmark-mode bundle. + */ +const HARNESS_IDENTIFIERS = [ + "__dockermapProbe", + "measureModelAcceptance", + "browserProbe", + "benchVite.config", + "benchAppVite.config", + "perf:time-to-answer", +"__dockermapBenchHelpers" + ,"__stage-five-control" + ,"stageFivePublicationControl" +]; + +/** + * The acceptance seam's identifiers. They are REAL product source (#335), but + * they must be compiled out of the shipped artifact and may only exist inside the + * seam module and the files that import it. + */ +const SEAM_IDENTIFIERS = [ + "__dockermapBenchAcceptanceSink", +"__dockermapBenchRenderDelayMs", +"__dockermapBenchDelayAfterNextAcceptance", +"__dockermapBenchDelayStartedAt", + "dockermapAcceptedRevision", + "recordModelAcceptance", + "useDeliveredModel", +"ModelAcceptanceStamp" +, +"__dockermapBenchLayerSink", +"__dockermapBenchFixtureGeneration", +"recordModelLayers" +]; + +const SEAM_MODULE = "apps/web/src/lib/performance/modelAcceptance.tsx"; +const SEAM_IMPORT = "lib/performance/modelAcceptance"; + +function walk(directory) { + const files = []; + for (const entry of readdirSync(directory)) { + const full = join(directory, entry); + const info = statSync(full); + if (info.isDirectory()) files.push(...walk(full)); + else files.push(full); + } + return files; +} + +function productionSources() { + const roots = ["apps/web/src", "apps/api/src", "packages/contracts/src"]; + return roots.flatMap((root) => { + const full = join(REPO_ROOT, root); + return existsSync(full) ? walk(full) : []; + }); +} + +function offendersIn(files, identifiers, filter = () => true) { + const offenders = []; + for (const file of files.filter(filter)) { + const body = readFileSync(file, "utf8"); + for (const identifier of identifiers) { + if (body.includes(identifier)) offenders.push(`${file.slice(REPO_ROOT.length + 1)} → ${identifier}`); + } + } + return offenders; +} + +describe("time-to-answer production isolation", () => { + it("ships no benchmark or seam identifier in the production web bundle", () => { + const dist = join(REPO_ROOT, "apps/web/dist"); + assert.ok( + existsSync(dist), + "apps/web/dist is missing: build the web app first (npm run build) so the shipped artifact can be inspected" + ); + const artifacts = walk(dist).filter((file) => /\.(js|mjs|css|html|json|map)$/.test(file)); + assert.ok(artifacts.length > 0, "the production build produced no inspectable artifacts"); + const offenders = offendersIn(artifacts, [...HARNESS_IDENTIFIERS, ...SEAM_IDENTIFIERS]); + assert.deepEqual( + offenders, + [], + `the production bundle references benchmark identifiers: ${offenders.join(", ")}` + ); + }); + + it("keeps harness identifiers entirely out of product sources", () => { + const offenders = offendersIn(productionSources(), HARNESS_IDENTIFIERS); + assert.deepEqual(offenders, [], `product sources reference harness identifiers: ${offenders.join(", ")}`); + }); + + it("confines the acceptance seam to the seam module and its import sites", () => { + const offenders = productionSources() + .filter((file) => file.slice(REPO_ROOT.length + 1) !== SEAM_MODULE) + .filter((file) => { + const body = readFileSync(file, "utf8"); + return SEAM_IDENTIFIERS.some((identifier) => body.includes(identifier)) && !body.includes(SEAM_IMPORT); + }) + .map((file) => file.slice(REPO_ROOT.length + 1)); + assert.deepEqual( + offenders, + [], + `a product source uses the acceptance seam without importing the seam module: ${offenders.join(", ")}` + ); + + const seam = readFileSync(join(REPO_ROOT, SEAM_MODULE), "utf8"); + // The seam must be gated by the compile-time flag, not by a runtime check, so + // it can be eliminated entirely from the product build. + assert.ok( + seam.includes("__DOCKERMAP_BENCH_ACCEPTANCE__"), + "the seam module must be gated by the compile-time __DOCKERMAP_BENCH_ACCEPTANCE__ flag" + ); + assert.ok( + !/if\s*\(\s*(?:import\.meta\.env|process\.env)/.test(seam), + "the seam must not be gated by a runtime environment check: it could not be eliminated" + ); + }); + + it("defines the compile-time seam flag as false in the production web config", () => { + const webConfig = readFileSync(join(REPO_ROOT, "apps/web/vite.config.ts"), "utf8"); + const definition = /__DOCKERMAP_BENCH_ACCEPTANCE__\s*:\s*([^,}\n]+)/.exec(webConfig); + assert.ok(definition, "the production web config must declare the seam flag"); + assert.equal( + definition[1].trim(), + '"false"', + "the production web config must define the seam flag as the literal string \"false\"" + ); + assert.ok(!webConfig.includes("benchVite"), "the production web config references the benchmark config"); + assert.ok(!webConfig.includes("benchApp"), "the production web config references the benchmark app build"); + }); + + it("does not reach a benchmark build config or entry from any production script", () => { + const pkg = JSON.parse(readFileSync(join(REPO_ROOT, "package.json"), "utf8")); + const productionScripts = Object.entries(pkg.scripts).filter( + ([name]) => !name.startsWith("perf:") && name !== "test:perf" + ); + const offenders = productionScripts.filter( + ([, command]) => + command.includes("benchVite") || + command.includes("benchAppVite") || + command.includes("tests/perf/capture") + ); + assert.deepEqual(offenders, [], `a production script invokes the benchmark: ${JSON.stringify(offenders)}`); + + const webPackage = JSON.parse(readFileSync(join(REPO_ROOT, "apps/web/package.json"), "utf8")); + for (const [name, command] of Object.entries(webPackage.scripts)) { + assert.ok( + !String(command).includes("bench") && !String(command).includes("tests/perf"), + `the production web script "${name}" reaches into the benchmark` + ); + } + }); + + it("compiles the seam into the benchmark-mode application build, not the product", () => { + const benchAppDist = join(REPO_ROOT, "tests/perf/.bench-app-dist"); + if (!existsSync(benchAppDist)) { + // The capture harness builds this and asserts the same property itself + // (fail closed) before it measures anything, so the check is not lost — it + // simply has no artifact to inspect outside a capture. + return; + } + const artifacts = walk(benchAppDist).filter((file) => /\.(js|mjs|html)$/.test(file)); + assert.ok(artifacts.length > 0, "the benchmark-mode build produced no inspectable artifacts"); + const compiled = artifacts.some((file) => + readFileSync(file, "utf8").includes("__dockermapBenchAcceptanceSink") + ); + assert.ok(compiled, "the benchmark-mode application build does not contain the acceptance seam"); + const offenders = offendersIn(artifacts, HARNESS_IDENTIFIERS); + assert.deepEqual( + offenders, + [], + `the benchmark-mode application build carries harness identifiers: ${offenders.join(", ")}` + ); + }); + + it("keeps the daemon stage attribution hook inert by default", () => { + const hookPath = join(REPO_ROOT, "crates/dockermap-daemon/src/bench_timing.rs"); + assert.ok(existsSync(hookPath), "the bench attribution hook is missing"); + const hook = readFileSync(hookPath, "utf8"); + assert.ok( + hook.includes("DOCKERMAP_BENCH_STAGE_TIMING_PATH"), + "the hook must be enabled by an explicit environment variable" + ); + assert.ok( + hook.includes("is_absolute()"), + "the hook must refuse a relative sink path, so it cannot silently write into a working directory" + ); + + // The hook must not be reachable from any route or API surface. + const offenders = []; + for (const file of productionSources()) { + const body = readFileSync(file, "utf8"); + if (body.includes("DOCKERMAP_BENCH_STAGE_TIMING_PATH")) offenders.push(file.slice(REPO_ROOT.length + 1)); + } + assert.deepEqual(offenders, [], `an API or web source reads the bench hook: ${offenders.join(", ")}`); + + // And it must only be consulted by the daemon's own collection paths. + const daemonSources = walk(join(REPO_ROOT, "crates/dockermap-daemon/src")).filter((file) => file.endsWith(".rs")); + const consumers = daemonSources + .filter((file) => readFileSync(file, "utf8").includes("bench_timing")) + .map((file) => file.slice(REPO_ROOT.length + 1)) + .sort(); + assert.deepEqual(consumers, [ + "crates/dockermap-daemon/src/cache_refresh.rs", + "crates/dockermap-daemon/src/main.rs" + ]); + }); + + it("adds no browser analytics or tracking to the production bundle", () => { + const dist = join(REPO_ROOT, "apps/web/dist"); + assert.ok(existsSync(dist), "apps/web/dist is missing: build the web app first"); + const analytics = [ + "google-analytics", + "googletagmanager", + "gtag(", + "sendBeacon", + "navigator.sendBeacon", + "mixpanel", + "posthog", + "sentry.init", + "datadogRum" + ]; + const offenders = offendersIn( + walk(dist).filter((entry) => /\.(js|mjs|html)$/.test(entry)), + analytics + ); + assert.deepEqual(offenders, [], `the production bundle carries analytics markers: ${offenders.join(", ")}`); + }); +}); diff --git a/tests/perf/stageFiveCaptureControl.mjs b/tests/perf/stageFiveCaptureControl.mjs new file mode 100644 index 00000000..3ce6250b --- /dev/null +++ b/tests/perf/stageFiveCaptureControl.mjs @@ -0,0 +1,12 @@ +/** + * Capture's Stage-5 control entrypoint (#335). + * + * This deliberately owns no protocol of its own. It gives capture its + * connection-before-trigger lifecycle while delegating arm/mark/trigger/ack to + * the same controller client exercised by the phase-control gate. + */ +import { armStageFivePublication } from "./stageFivePublicationControl.mjs"; + +export function armCaptureStageFivePublication(input) { + return armStageFivePublication(input); +} diff --git a/tests/perf/stageFivePublicationControl.mjs b/tests/perf/stageFivePublicationControl.mjs new file mode 100644 index 00000000..7e907252 --- /dev/null +++ b/tests/perf/stageFivePublicationControl.mjs @@ -0,0 +1,176 @@ +/** + * Benchmark-only daemon proxy for the Stage-5 phase sweep. It is deliberately + * in tests/perf: product processes never import, start, or route to it. + * + * The proxy learns the API poller from its real /daemon/health requests. Once + * the fixture-triggered daemon revision is witnessed, it keeps that exact + * response unavailable until the requested offset after an observed poll, then + * makes it available to the following real poll. Thus the controller changes + * neither the API timer nor the daemon; it only supplies a deterministic test + * fixture publication barrier. + */ +import { createServer, request as httpRequest } from "node:http"; + +const json = (response, status, body) => { + const payload = JSON.stringify(body); + response.writeHead(status, { "content-type": "application/json", "content-length": Buffer.byteLength(payload) }); + response.end(payload); +}; + +function upstreamGet(origin, path) { + return new Promise((resolve, reject) => { + const target = new URL(path, origin); + const call = httpRequest(target, { method: "GET" }, (response) => { + let body = ""; + response.on("data", (chunk) => (body += chunk)); + response.on("end", () => { + if ((response.statusCode ?? 500) >= 400) reject(new Error(`upstream ${response.statusCode} ${path}`)); + else resolve({ status: response.statusCode ?? 200, headers: response.headers, body }); + }); + }); + call.on("error", reject); + call.end(); + }); +} + +/** + * The only client protocol for a controlled Stage-5 publication. Keeping this + * beside the test-only controller makes the runnable phase gate and the full + * capture use the identical arm -> mark -> trigger -> exact-ack mechanism. + * + * `armStageFivePublication()` intentionally does not release immediately: + * capture must connect its real API-SSE observer after arming but before the + * trigger. `release()` marks first, so no triggered publication can enter the + * controller's forbidden unmarked window. + */ +async function postControl(url, body) { + const response = await fetch(url, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify(body) + }); + const payload = await response.json(); + if (!response.ok) throw new Error(`stage-5 publication controller rejected ${url}: ${JSON.stringify(payload)}`); + return payload; +} + +export async function armStageFivePublication({ controllerUrl, triggerId, requestedPhaseMs, previousRevision, timeoutMs, withholdUntilOpen = false }) { + await postControl(`${controllerUrl}/__stage-five-control/arm`, { triggerId, requestedPhaseMs, previousRevision, withholdUntilOpen }); + return { + async release(trigger) { + await postControl(`${controllerUrl}/__stage-five-control/mark`, { triggerId }); + await trigger(); + const deadline = Date.now() + timeoutMs; + while (Date.now() < deadline) { + const response = await fetch(`${controllerUrl}/__stage-five-control/ack`); + if (response.status === 200) { + const ack = await response.json(); + if (ack.triggerId !== triggerId) throw new Error("stage-5 publication acknowledgement has the wrong trigger identity"); + if (ack.revision === previousRevision) throw new Error("stage-5 publication acknowledgement did not identify a new revision"); + return ack; + } + if (response.status >= 400) throw new Error(`stage-5 publication controller rejected ${triggerId}: ${await response.text()}`); + await new Promise((done) => setTimeout(done, 2)); + } + throw new Error(`stage-5 publication controller did not acknowledge ${triggerId}`); + }, + async open() { + await postControl(`${controllerUrl}/__stage-five-control/open`, { triggerId }); + } + }; +} + +export function startStageFivePublicationController({ upstream, port = 0, now = () => performance.now() }) { + let stale = null; + let armed = null; + let failure = null; + let timer = null; + + function fail(message) { + failure = message; + if (armed) armed.failure = message; + } + function arm(body) { + if (armed && !armed.ack) throw new Error("a stage-5 publication is already armed"); + if (!body?.triggerId || !Number.isFinite(body.requestedPhaseMs) || !body.previousRevision) { + throw new Error("arm requires triggerId, requestedPhaseMs and previousRevision"); + } + failure = null; + armed = { triggerId: String(body.triggerId), requestedPhaseMs: Number(body.requestedPhaseMs), previousRevision: String(body.previousRevision), withholdUntilOpen: Boolean(body.withholdUntilOpen), opened: !body.withholdUntilOpen, marked: false, exact: null, pollAtMs: 0, releasedAtMs: 0, ack: null, failure: null }; + return armed; + } + function open(body) { + if (!armed || armed.triggerId !== body?.triggerId || !armed.ack) throw new Error("the opened trigger has no acknowledged publication"); + armed.opened = true; + return armed; + } + function mark(body) { + if (!armed || armed.triggerId !== body?.triggerId) throw new Error("the marked trigger is not the armed stage-5 publication"); + if (armed.failure) throw new Error(armed.failure); + armed.marked = true; + return armed; + } + async function health(response) { + const upstreamResponse = await upstreamGet(upstream, "/daemon/health"); + const revision = JSON.parse(upstreamResponse.body).modelRevision; + if (!stale) stale = upstreamResponse; + if (!armed) return writeProxy(response, upstreamResponse); + if (armed.failure) return json(response, 409, { error: armed.failure }); + if (revision !== armed.previousRevision && !armed.exact) { + if (!armed.marked) { + fail(`wrong publication ${revision} arrived before trigger ${armed.triggerId} was marked`); + return json(response, 409, { error: failure }); + } + armed.exact = { revision, response: upstreamResponse }; + // A changed response belongs to the trigger only after mark(). It is + // retained, never substituted by a later revision. + } else if (armed.exact && revision !== armed.previousRevision && revision !== armed.exact.revision) { + fail(`wrong publication ${revision} followed trigger ${armed.triggerId}; exact revision is ${armed.exact.revision}`); + return json(response, 409, { error: failure }); + } + if (armed.exact && !armed.pollAtMs) { + armed.pollAtMs = now(); + timer = setTimeout(() => { + if (!armed || armed.failure || !armed.exact) return; + armed.releasedAtMs = now(); + armed.ack = { triggerId: armed.triggerId, revision: armed.exact.revision, requestedPhaseMs: armed.requestedPhaseMs, pollAtMs: armed.pollAtMs, releasedAtMs: armed.releasedAtMs }; + }, armed.requestedPhaseMs); + return writeProxy(response, stale); + } + if (armed.ack && armed.opened) return writeProxy(response, armed.exact.response); + return writeProxy(response, stale); + } + function writeProxy(response, proxied) { + response.writeHead(proxied.status, proxied.headers); + response.end(proxied.body); + } + const server = createServer(async (request, response) => { + try { + if (request.url === "/__stage-five-control/arm" && request.method === "POST") { + let text = ""; + for await (const chunk of request) text += chunk; + json(response, 200, arm(JSON.parse(text || "{}"))); + } else if (request.url === "/__stage-five-control/mark" && request.method === "POST") { + let text = ""; + for await (const chunk of request) text += chunk; + json(response, 200, mark(JSON.parse(text || "{}"))); + } else if (request.url === "/__stage-five-control/open" && request.method === "POST") { + let text = ""; + for await (const chunk of request) text += chunk; + json(response, 200, open(JSON.parse(text || "{}"))); + } else if (request.url === "/__stage-five-control/ack") { + if (!armed) json(response, 404, { error: "no armed stage-5 publication" }); + else if (armed.failure) json(response, 409, { error: armed.failure }); + else if (!armed.ack) json(response, 202, { pending: true }); + else json(response, 200, armed.ack); + } else if (request.url?.startsWith("/daemon/health")) await health(response); + else writeProxy(response, await upstreamGet(upstream, request.url ?? "/")); + } catch (error) { + json(response, 500, { error: String(error) }); + } + }); + return new Promise((resolve) => server.listen(port, "127.0.0.1", () => resolve({ + url: `http://127.0.0.1:${server.address().port}`, + close: () => new Promise((done) => { if (timer) clearTimeout(timer); server.close(done); }) + }))); +} diff --git a/tests/perf/stageFivePublicationControl.test.mjs b/tests/perf/stageFivePublicationControl.test.mjs new file mode 100644 index 00000000..4800f6c0 --- /dev/null +++ b/tests/perf/stageFivePublicationControl.test.mjs @@ -0,0 +1,48 @@ +import assert from "node:assert/strict"; +import { createServer } from "node:http"; +import test from "node:test"; +import { armStageFivePublication, startStageFivePublicationController } from "./stageFivePublicationControl.mjs"; + +async function upstream() { + let revision = "before"; + const server = createServer((_request, response) => response.end(JSON.stringify({ modelRevision: revision }))); + await new Promise((done) => server.listen(0, "127.0.0.1", done)); + return { url: `http://127.0.0.1:${server.address().port}`, set: (value) => (revision = value), close: () => new Promise((done) => server.close(done)) }; +} +async function post(url, body) { return fetch(url, { method: "POST", headers: { "content-type": "application/json" }, body: JSON.stringify(body) }); } + +test("shared release helper marks before it triggers and returns the exact acknowledged publication", async () => { + const daemon = await upstream(); + const control = await startStageFivePublicationController({ upstream: daemon.url }); + try { + assert.equal((await fetch(`${control.url}/daemon/health`)).status, 200); + const publication = await armStageFivePublication({ + controllerUrl: control.url, + triggerId: "generation-1", + requestedPhaseMs: 20, + previousRevision: "before", + timeoutMs: 1_000 + }); + let triggered = false; + const ackPromise = publication.release(() => { triggered = true; daemon.set("triggered"); }); + assert.equal((await (await fetch(`${control.url}/daemon/health`)).json()).modelRevision, "before"); + const ack = await ackPromise; + assert.equal(triggered, true); + assert.equal(ack.triggerId, "generation-1"); + assert.equal(ack.revision, "triggered"); + assert.equal((await (await fetch(`${control.url}/daemon/health`)).json()).modelRevision, "triggered"); + } finally { await control.close(); await daemon.close(); } +}); + +test("wrong publication before the marked trigger is RED", async () => { + const daemon = await upstream(); + const control = await startStageFivePublicationController({ upstream: daemon.url }); + try { + await fetch(`${control.url}/daemon/health`); + await post(`${control.url}/__stage-five-control/arm`, { triggerId: "generation-2", requestedPhaseMs: 20, previousRevision: "before" }); + daemon.set("wrong-publication"); + assert.equal((await fetch(`${control.url}/daemon/health`)).status, 409); + const ack = await fetch(`${control.url}/__stage-five-control/ack`); + assert.equal(ack.status, 409); + } finally { await control.close(); await daemon.close(); } +}); diff --git a/tests/perf/stageSixSevenIndependenceIsolation.test.mjs b/tests/perf/stageSixSevenIndependenceIsolation.test.mjs new file mode 100644 index 00000000..00218a8f --- /dev/null +++ b/tests/perf/stageSixSevenIndependenceIsolation.test.mjs @@ -0,0 +1,50 @@ +/** The seam-isolation control is supporting evidence, isolated from Baseline-4. */ +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { resolve } from "node:path"; +import test from "node:test"; + +const root = resolve(new URL("../..", import.meta.url).pathname); +const read = (path) => readFileSync(resolve(root, path), "utf8"); + +test("normal Stage-6/7 capture and calibration cannot contain injected-delay samples", () => { +const capture = read("tests/perf/capture.ts"); +assert.doesNotMatch(capture, /BenchRenderDelayMs/); +assert.doesNotMatch(capture, /TIME_TO_ANSWER_INDEPENDENCE_(?:DELAY|SAMPLES)/); +assert.doesNotMatch(capture, /assertStageSixSevenIndependence/); +assert.doesNotMatch(capture, /controlStage(?:Six|Seven)Ms/); +}); + +test("seam isolation is a dedicated supporting protocol, never a silent capture mode", () => { +const pkg = JSON.parse(read("package.json")); +const protocol = read("tests/perf/captureIndependence.ts"); +assert.equal(pkg.scripts["perf:independence"], "tsx tests/perf/captureIndependence.ts"); +assert.match(protocol, /controlled-stage6-stage7-seam-isolation/); +assert.doesNotMatch(protocol, /from ["']\.\/capture(?:\.ts)?["']/); +assert.match(protocol, /armSeamIsolationDelay/); +assert.match(protocol, /assertStageSixSevenIndependence/); +}); + +test("the private seam control retains real acceptance, coherence, post-acceptance delay and isolation", () => { +const protocol = read("tests/perf/captureIndependence.ts"); +for (const primitive of ["fake-docker-api.mjs", "apps/api/src/index.ts", "chromium.launch", "startStaticServer", "finally", "rmSync(work", "acceptedRevision !== measured.snapshotRevision", "delayStartedAfterAcceptance"]) assert.match(protocol, new RegExp(primitive.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"))); +assert.doesNotMatch(protocol, /startStageFivePublicationController/); +assert.doesNotMatch(protocol, /armStageFivePublication/); +assert.doesNotMatch(protocol, /setExpectedModelRevision/); +assert.doesNotMatch(protocol, /triggerRevision/); +}); + +test("composite rejects publication-attribution-shaped control artifacts", () => { +const assembler = read("tests/perf/assembleCompositeEvidence.ts"); +assert.match(assembler, /controlled-stage6-stage7-seam-isolation/); +assert.match(assembler, /publication-level causal identity unavailable/); +assert.match(assembler, /validatesDaemonToBrowserAttribution !== false/); +assert.match(assembler, /"triggerRevision" in sample/); +assert.match(read("tests/perf/captureIndependence.ts"), /controlSamplesExcludedFromBaselineTiming/); +}); + +test("seam isolation is supporting evidence rather than a Baseline-4 prerequisite", () => { +const assembler = read("tests/perf/assembleCompositeEvidence.ts"); +assert.match(assembler, /stageSixSeven === null\) return validateTimeToAnswerEvidence/); +assert.match(assembler, /supporting-evidence\.json/); +}); diff --git a/tests/perf/staticServer.mjs b/tests/perf/staticServer.mjs new file mode 100644 index 00000000..d3819c57 --- /dev/null +++ b/tests/perf/staticServer.mjs @@ -0,0 +1,80 @@ +/** + * Minimal deterministic static file server for the benchmark (#335). + * + * It serves a built directory on a private port with SPA history fallback. It + * exists so the capture harness does not depend on a dev server's caching, + * HMR, or transform behaviour: the numbers must come from built artifacts. + */ +import { createServer } from "node:http"; +import { readFile, stat } from "node:fs/promises"; +import { extname, join, normalize, resolve } from "node:path"; + +const CONTENT_TYPES = { + ".html": "text/html; charset=utf-8", + ".js": "text/javascript; charset=utf-8", + ".mjs": "text/javascript; charset=utf-8", + ".css": "text/css; charset=utf-8", + ".json": "application/json; charset=utf-8", + ".svg": "image/svg+xml", + ".png": "image/png", + ".ico": "image/x-icon", + ".woff2": "font/woff2" +}; + +export async function startStaticServer({ directory, port, host = "127.0.0.1" }) { + const root = resolve(directory); + const server = createServer(async (request, response) => { + const requested = (request.url ?? "/").split("?")[0]; + const decoded = decodeURIComponent(requested); + const relative = normalize(decoded).replace(/^(\.\.[/\\])+/, ""); + let target = join(root, relative); + // Path traversal must never escape the served root. + if (!target.startsWith(root)) target = join(root, "index.html"); + try { + const info = await stat(target); + if (info.isDirectory()) target = join(target, "index.html"); + } catch { + target = join(root, "index.html"); + } + try { + const body = await readFile(target); + response.writeHead(200, { + "content-type": CONTENT_TYPES[extname(target)] ?? "application/octet-stream", + "content-length": body.byteLength, + // No caching: every benchmark run must load the same bytes from cold. + "cache-control": "no-store" + }); + response.end(body); + } catch { + response.writeHead(404, { "content-type": "text/plain" }); + response.end("not found"); + } + }); + await new Promise((done, fail) => { + server.once("error", fail); + server.listen(port ?? 0, host, done); + }); + // Report the port the OS actually bound. Callers that pass 0 get an atomic + // allocation instead of the reserve-then-bind race of reservePort(), where an + // unrelated process (or a previous child not yet reaped) can take the port in + // between — which aborts an expensive capture with EADDRINUSE. + const address = server.address(); + const boundPort = address && typeof address === "object" ? address.port : port; + return { + port: boundPort, + url: `http://${host}:${boundPort}`, + async close() { + await new Promise((done) => server.close(done)); + } + }; +} + +/** Ask the OS for a free port so parallel benchmark runs cannot collide. */ +export async function reservePort() { + const { createServer: create } = await import("node:net"); + const probe = create(); + await new Promise((done) => probe.listen(0, "127.0.0.1", done)); + const { port } = probe.address(); + await new Promise((done) => probe.close(done)); + return port; +} diff --git a/tests/perf/summarize.ts b/tests/perf/summarize.ts new file mode 100644 index 00000000..665d258f --- /dev/null +++ b/tests/perf/summarize.ts @@ -0,0 +1,111 @@ +#!/usr/bin/env node +/** + * Recompute and print the time-to-answer summary from a closed artifact (#335). + * + * npm run perf:summarize -- --artifact [--markdown] + * + * Summaries are ALWAYS recomputed from the raw samples here; the artifact never + * carries them. This is the same path a reviewer uses, so the numbers in the + * interpretation document cannot drift from the evidence. + */ +import { readFileSync } from "node:fs"; +import { + TIME_TO_ANSWER_STAGES, + derivedTimeToAnswerPhaseNormalized, + derivedTimeToAnswerSummaries, + validateTimeToAnswerEvidence +} from "../../apps/web/src/lib/performance/timeToAnswerEvidence"; +import { isPhaseControlledFixture } from "../../apps/web/src/lib/performance/timeToAnswerPollPhase"; + +const args = Object.fromEntries( + process.argv + .slice(2) + .flatMap((value, index, all) => (value.startsWith("--") ? [[value.slice(2), all[index + 1] ?? ""]] : [])) +); +const artifactPath = args.artifact; +if (!artifactPath || artifactPath.startsWith("--")) { + throw new Error("Usage: npm run perf:summarize -- --artifact "); +} + +const evidence = validateTimeToAnswerEvidence(JSON.parse(readFileSync(artifactPath, "utf8"))); +const summaries = derivedTimeToAnswerSummaries(evidence); +const stageById = new Map(TIME_TO_ANSWER_STAGES.map((stage) => [stage.id, stage])); + +const fixtures = [...new Set(evidence.records.map((record) => record.fixture))]; +const order = [...new Set(evidence.records.map((record) => record.stage))]; + +const rows: string[] = []; +rows.push("| fixture | stage | run p95 (ms) | reviewed aggregation | reviewed (ms) | min | max |"); +rows.push("| --- | --- | --- | --- | --- | --- |"); +for (const fixture of fixtures) { + for (const stage of order) { + const summary = summaries.get(`${fixture}\u0000${stage}`); + if (!summary) continue; + const record = evidence.records.find((entry) => entry.fixture === fixture && entry.stage === stage)!; + const all = record.runs.flat(); + rows.push( + `| ${fixture} | ${stage} | ${summary.runP95Ms.map((value) => value.toFixed(2)).join(" / ")} | ${summary.reviewedAggregation} | ${summary.reviewedMs.toFixed(2)} | ${Math.min(...all).toFixed(2)} | ${Math.max(...all).toFixed(2)} |` + ); + } +} + +process.stdout.write(`${rows.join("\n")}\n\n`); + +// Stage 5: the declared-phase curve and the phase-normalized figure. Both are +// recomputed here from the raw samples, using the declared grid — never read from +// the artifact, which stores raw numbers only. +const stageFiveFixtures = fixtures.filter( + (fixture) => + isPhaseControlledFixture(fixture) && + evidence.records.some( + (record) => record.fixture === fixture && record.stage === "publicationToNodeObservationMs" + ) +); +if (stageFiveFixtures.length > 0) { + const intervalMs = Number(evidence.environment.ssePollIntervalMs); + process.stdout.write( + "### stage 5 — publication → Node observation, by declared poll phase\n\n" + + "| fixture | declared phases | observed latency median per declared phase (ms, earliest→latest) | phase-normalized p95 (ms) | span (ms) |\n" + + "| --- | --- | --- | --- | --- |\n" + ); + for (const fixture of stageFiveFixtures) { + const record = evidence.records.find( + (entry) => entry.fixture === fixture && entry.stage === "publicationToNodeObservationMs" + )!; + const normalized = derivedTimeToAnswerPhaseNormalized(record.runs, evidence.environment.ssePollIntervalMs); + const flat = record.runs.flat(); + process.stdout.write( + `| ${fixture} | ${normalized.phaseMediansMs.length} | ${normalized.phaseMediansMs + .map((value) => value.toFixed(0)) + .join(", ")} | ${normalized.phaseNormalizedP95Ms.toFixed(2)} | ${(Math.max(...flat) - Math.min(...flat)).toFixed(2)} |\n` + ); + } + process.stdout.write( + "\nThe phase-normalized figure weights the DECLARED phases uniformly to characterise the latency the fixed " + + `${intervalMs} ms polling mechanism imposes. It is not an observed user-traffic distribution and not network latency.\n\n` + ); +} + +// Bucket roll-up per reference fixture: where the time actually goes. +const referenceFixtures = fixtures.filter((fixture) => fixture.startsWith("reference-")); +for (const fixture of referenceFixtures) { + const bucketTotals = new Map(); + let total = 0; + for (const stage of TIME_TO_ANSWER_STAGES) { + const summary = summaries.get(`${fixture}\u0000${stage.id}`); + if (!summary) continue; + const definition = stageById.get(stage.id)!; + bucketTotals.set(definition.bucket, (bucketTotals.get(definition.bucket) ?? 0) + summary.reviewedMs); + total += summary.reviewedMs; + } + process.stdout.write(`\n### ${fixture} buckets (sum of reviewed stage figures: ${total.toFixed(2)} ms)\n`); + for (const [bucket, value] of [...bucketTotals.entries()].sort((left, right) => right[1] - left[1])) { + process.stdout.write( + `- ${bucket}: ${value.toFixed(2)} ms (${((value / total) * 100).toFixed(1)}%)\n` + ); + } +} + +process.stdout.write( + `\nenvironment: ${JSON.stringify(evidence.environment, null, 2)}\n` +); diff --git a/tests/perf/tsconfig.json b/tests/perf/tsconfig.json new file mode 100644 index 00000000..1b47df4c --- /dev/null +++ b/tests/perf/tsconfig.json @@ -0,0 +1,8 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { + "types": ["node"], + "allowJs": true + }, + "include": ["capture.ts", "captureIndependence.ts", "summarize.ts", "preconditioning.ts", "phaseControl.ts"] +}