Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
81 commits
Select commit Hold shift + click to select a range
879b0d8
feat: add the closed time-to-answer evidence contract (#335)
Joncallim Sep 22, 2026
4473423
feat: add the deterministic Docker fixture source for time-to-answer …
Joncallim Sep 22, 2026
985be7b
feat: add inert bench-only stage attribution to the daemon (#335)
Joncallim Sep 22, 2026
23ed490
docs: record the bench attribution hook and the remaining #335 work i…
Joncallim Sep 22, 2026
b6904d5
feat: add the time-to-answer capture command and browser probes (#335)
Joncallim Sep 23, 2026
71494a0
feat: capture and interpret the first time-to-answer baseline (#335)
Joncallim Sep 23, 2026
0714c87
fix: remediate the #335 review findings before recapturing the baseli…
Joncallim Sep 23, 2026
c0baa05
docs: rewrite the baseline interpretation from the corrected capture …
Joncallim Sep 23, 2026
c622da2
fix: cold/warm stage split and daemon binary provenance (#335, round …
Joncallim Sep 23, 2026
323c851
docs: correct the evidence and baseline docs to match the executable …
Joncallim Sep 23, 2026
597566e
docs: state plainly that no accepted baseline exists yet (#335)
Joncallim Sep 23, 2026
dff156e
feat: instrument the real coherent-model acceptance seam (#335)
Joncallim Sep 23, 2026
f9dcff5
fix: build the pinned daemon with the workspace manifest path (#335)
Joncallim Sep 23, 2026
491a019
fix: harden stage-6/7 arming and API revision attribution (#335)
Joncallim Sep 23, 2026
7171234
fix: make stage 5 and stages 6/7 sample-scoped observers (#335)
Joncallim Sep 23, 2026
9a8ca40
fix: hold the stage-5 observation open until it sees the accepted rev…
Joncallim Sep 23, 2026
4cccf0a
feat: retain the warm-up observation window as auditable evidence (#335)
Joncallim Sep 23, 2026
5bb59a9
fix: give stage 6 an acceptance-only mode for provider-state fixtures…
Joncallim Sep 23, 2026
aa26a80
fix: attribute stage 6 to the fetch cycle that delivered the accepted…
Joncallim Sep 23, 2026
cf77e8b
fix: remove the port-reservation race from the capture harness (#335)
Joncallim Sep 23, 2026
16abe26
docs: record baseline 3 as the time-to-answer measurement authority (…
Joncallim Sep 23, 2026
ab75c6e
feat!: revise the time-to-answer methodology to a deterministic stage…
Joncallim Sep 23, 2026
6862f30
fix: relax the stage-5 phase-coverage minimum only for probing runs (…
Joncallim Sep 23, 2026
a37f893
fix: fit the daemon publication grid by least squares and track it fr…
Joncallim Sep 23, 2026
2b7d508
fix: fit the stage-5 grid over refresh cycles, not every revision cha…
Joncallim Sep 23, 2026
5f1981d
fix: attribute each stage-5 sample to the publication its poll tick a…
Joncallim Sep 23, 2026
252943e
fix: build the stage-5 grid from the daemon's own cycle timestamps (#…
Joncallim Sep 23, 2026
dfde604
feat: split stage-5 cells into phase-controlled and free-running (#335)
Joncallim Sep 23, 2026
33ce32e
fix: exclude the connect frame from a stage-5 sample's observation wi…
Joncallim Sep 23, 2026
bc40917
fix: coarsen the stage-5 grid to the precision the poller actually ha…
Joncallim Sep 23, 2026
08853f5
test: carry notification and acceptance state in the stage-7 failure …
Joncallim Sep 23, 2026
021e47d
test: confirm fixture triggers land, rather than trusting the POST (#…
Joncallim Sep 23, 2026
8cba3b9
test: confirm the control trigger reaches fixture, daemon and API (#335)
Joncallim Sep 23, 2026
6a66a65
fix(bench): aggregate every Stage-5 phase observation and reconcile t…
Joncallim Sep 23, 2026
030d460
docs(bench): record the free-running Stage-5 cells and pin the exclusion
Joncallim Sep 23, 2026
942f205
test(bench): check provider cells remain free-running
Joncallim Sep 23, 2026
5b0b285
test(bench): assert phase-normalized exclusion wording
Joncallim Sep 23, 2026
b3e990e
fix(bench): make the stage-7 control samples observe the publication …
Joncallim Sep 23, 2026
bcf13fe
test(bench): cover the stage-7 trigger fence
Joncallim Sep 23, 2026
18bc540
test(bench): harden the stage-7 trigger fence proof
Joncallim Sep 23, 2026
5e2c3c9
fix(bench): bind stage-7 control to coherent publication
Joncallim Sep 23, 2026
e2ba50e
fix(bench): gate control until revision is known
Joncallim Sep 23, 2026
c92d1da
test(bench): cover revision-gate ordering
Joncallim Sep 23, 2026
1213bf9
fix(bench): retain fixed-ten warm-up evidence
Joncallim Sep 23, 2026
103240a
fix(bench): isolate browser capture runs
Joncallim Sep 23, 2026
fb726e1
fix(bench): repair capture syntax and guard harness compilation
Joncallim Sep 23, 2026
1952f11
fix(bench): restore capture entrypoint and type-check the harness
Joncallim Sep 23, 2026
6d00071
fix(bench): close the capture entrypoint brace
Joncallim Sep 23, 2026
30df65f
fix(bench): restore capture entrypoint scope
Joncallim Sep 23, 2026
65785ad
fix(bench): close stage-five observer scope
Joncallim Sep 23, 2026
bb9409e
test(bench): record model layers and browser lifecycle for diagnosis
Joncallim Sep 24, 2026
d6fd25f
fix(bench): witness publication phase and lifecycle cleanup
Joncallim Sep 24, 2026
1ae6dd4
test(bench): add sustained preconditioning gate
Joncallim Sep 24, 2026
0afaf5a
test(bench): enforce preconditioning depth
Joncallim Sep 24, 2026
a5930b7
fix(bench): control stage-five publication phase
Joncallim Sep 24, 2026
b1f0453
fix(bench): share stage-five capture control
Joncallim Sep 25, 2026
2347271
test(bench): exercise capture stage-five control
Joncallim Sep 25, 2026
10c224e
test(bench): pin capture phase-control entrypoint
Joncallim Sep 25, 2026
179bb15
feat(bench): add composite stage-five evidence
Joncallim Sep 25, 2026
ab0b118
feat(perf): add frozen warm-up calibration guard
Joncallim Sep 25, 2026
8f99a51
test(perf): guard calibrated warm-up protocol
Joncallim Sep 25, 2026
551b52f
fix(perf): collect warm-up calibration evidence
Joncallim Sep 25, 2026
c15fecc
fix(perf): support calibration diagnostics
Joncallim Sep 25, 2026
d254152
fix(perf): isolate calibration phase collection
Joncallim Sep 25, 2026
41d95c4
fix(perf): skip baseline control during calibration
Joncallim Sep 25, 2026
71fa72d
refactor(perf): isolate stage six seven controls
Joncallim Sep 25, 2026
556db90
test(perf): cover controlled independence provenance
Joncallim Sep 25, 2026
9a03888
fix(perf): revise calibration window methodology
Joncallim Sep 26, 2026
56c8e85
feat(perf): add independence protocol entrypoint
Joncallim Sep 26, 2026
edeee18
fix(perf): derive timing ownership from protocols
Joncallim Sep 26, 2026
2db0dd1
fix(perf): retain complete calibration conflict reports
Joncallim Sep 26, 2026
882b7e3
feat(perf): fix Baseline-4 burn-in window
Joncallim Sep 26, 2026
225429e
fix(perf): retain burn-in only for warmed stages
Joncallim Sep 26, 2026
74d8249
feat(perf): self-orchestrate independence control
Joncallim Sep 26, 2026
bff4ec7
fix(perf): bound independence fixture generation
Joncallim Sep 26, 2026
13d02ab
fix(perf): bind independence to accepted model pair
Joncallim Sep 26, 2026
1f6f1af
fix(perf): gate acceptance until publication release
Joncallim Sep 26, 2026
7722a79
fix(perf): rescope stage seam isolation
Joncallim Sep 26, 2026
bdce6ae
fix(perf): record acceptance delay provenance
Joncallim Sep 26, 2026
67563c0
docs(perf): publish Baseline-4 authority and reconcile methodology-8 …
Joncallim Sep 27, 2026
1fae54a
docs(perf): correct the seam-isolation protocol description (#335)
Joncallim Sep 27, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -13,3 +13,5 @@ target
.codex/*
!.codex/agents/
!.codex/agents/*.toml
tests/perf/.bench-dist
tests/perf/.bench-app-dist
12 changes: 12 additions & 0 deletions apps/web/src/bench-flag.d.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,12 @@
/**
* Compile-time flag for the benchmark-only instrumentation seam (#335).
*
* The ordinary production build defines it `false` (apps/web/vite.config.ts), so
* every benchmark branch is removed by dead-code elimination before
* minification. The benchmark-mode application build in `tests/perf` defines it
* `true`.
*
* It is declared here so BOTH builds typecheck against the same product source:
* the seam lives in real application code, never in a copied implementation.
*/
declare const __DOCKERMAP_BENCH_ACCEPTANCE__: boolean;
9 changes: 9 additions & 0 deletions apps/web/src/components/AppShell.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@ import { AppContext } from "../context";
import Icon, { type IconName } from "./Icon";
import CommandPalette from "./CommandPalette";
import RouteFocusManager from "./RouteFocusManager";
import { ModelAcceptanceStamp } from "../lib/performance/modelAcceptance";
import { StateDot, Tag } from "./primitives";
import { UNAVAILABLE_USER } from "../lib/identity";

Expand Down Expand Up @@ -275,6 +276,14 @@ export default function AppShell({ onBearerSignOut }: { onBearerSignOut: () => v
<RouteFocusManager />
</div>

{/*
Benchmark-only acceptance stamp (#335). It renders nothing and has no
effect in the product build; the benchmark application build stamps the
accepted revision token in the same commit that renders the accepted
model, so the capture can attribute a DOM repaint to a revision.
*/}
<ModelAcceptanceStamp revision={model?.modelRevision ?? null} />

<CommandPalette open={commandOpen} onClose={() => setCommandOpen(false)} model={model} />
</AppContext.Provider>
);
Expand Down
24 changes: 20 additions & 4 deletions apps/web/src/hooks/useSystemModel.ts
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@ import { projectRuntimeMap } from "../lib/atlas/project";
import type { AtlasEnvelope } from "../lib/atlas/types";
import type { EvidenceMode, ModelProvenance } from "../lib/evidence";
import { modelProvenanceForMode } from "../lib/evidence";
import { recordModelAcceptance, recordModelLayers, useDeliveredModel } from "../lib/performance/modelAcceptance";
import { useApiResource } from "./useApiResource";

export interface SystemModelState {
Expand Down Expand Up @@ -62,6 +63,15 @@ export function useSystemModel(refreshTick: number, evidenceMode: EvidenceMode |
const built = buildModel(snapshot.data, runtimeMap.data);
lastModel.current = built;
lastProvenance.current = snapshot.provenance;
// The acceptance seam (#335). This is the exact point at which a fetched
// resource/revision pair BECOMES the coherent model the UI renders, so it is
// where the benchmark's stage-6 clock starts. It is compile-time gated and
// carries only an opaque timestamp + revision token; see
// lib/performance/modelAcceptance.tsx.
if (__DOCKERMAP_BENCH_ACCEPTANCE__) {
recordModelAcceptance(snapshot.data.modelRevision, runtimeMap.data.modelRevision);
recordModelLayers(snapshot.data, runtimeMap.data, built);
}
return built;
}, [snapshot.data, snapshot.generation, snapshot.provenance, runtimeMap.data, runtimeMap.generation, runtimeMap.provenance]);

Expand Down Expand Up @@ -103,10 +113,16 @@ export function useSystemModel(refreshTick: number, evidenceMode: EvidenceMode |
}, [snapshot.data, snapshot.generation, snapshot.provenance, runtimeMap.data, runtimeMap.generation, runtimeMap.provenance]);

return {
model,
atlas,
findings,
modelProvenance,
/**
* The publication seam (#335). In the product build `useDeliveredModel` is
* the identity function — no state, no effects, no behavioural difference.
* The benchmark application build defines the compile-time flag, which is
* where the artificial presentation-delay control withholds a newly accepted
* publication from the render tree. One publication (model + atlas +
* findings + provenance) is delayed as a unit so the render tree can never
* observe a split state.
*/
...useDeliveredModel({ model, atlas, findings, modelProvenance }, model?.modelRevision ?? null),
loading: snapshot.loading || runtimeMap.loading,
error: snapshot.error ?? runtimeMap.error
};
Expand Down
197 changes: 197 additions & 0 deletions apps/web/src/lib/performance/modelAcceptance.tsx
Original file line number Diff line number Diff line change
@@ -0,0 +1,197 @@
/**
* Benchmark-only instrumentation seam for the time-to-answer baseline (#335).
*
* This is REAL production application code, not a copy: `useSystemModel` calls
* `recordModelAcceptance()` at the exact point where a freshly fetched
* resource/revision pair becomes the coherent model the UI renders, and routes
* that publication through `useDeliveredModel()`. Both are gated by the
* compile-time constant `__DOCKERMAP_BENCH_ACCEPTANCE__`:
*
* - production (`apps/web/vite.config.ts`) defines it `false`, so the whole
* module collapses to `return value` / `return null` and every benchmark
* branch, event identifier and delay mechanism is eliminated from the shipped
* bundle (`tests/perf/productionIsolation.test.mjs` inspects the artifact);
* - the benchmark-mode application build in `tests/perf` defines it `true`,
* which is how the capture observes the real acceptance seam and how the
* stage-6/7 independence control injects an artificial presentation delay
* *after* acceptance.
*
* The seam carries no product payload and no telemetry. It records an opaque
* timing event plus the model revision token that was accepted, in an in-memory
* sink on the page, and (benchmark build only) stamps that same opaque token on
* the document root so the capture can prove the DOM content it times belongs to
* the accepted revision's render. Nothing leaves the page; nothing is uploaded.
*/
import { useEffect, useLayoutEffect, useRef, useState, type ReactElement } from "react";
import type { DockerSnapshot, RuntimeMap } from "@dockermap/contracts";
import { summarize, type SystemModel } from "../model";

/** One accepted coherent model: opaque timings only. */
export interface ModelAcceptanceEvent {
/** Monotonic sequence number within this page. */
seq: number;
/** `performance.now()` at the instant the coherent model was accepted. */
at: number;
/** The opaque daemon model revision token the accepted model belongs to. */
revision: string;
/** The exact snapshot/runtime-map pair consumed by useSystemModel. */
snapshotRevision: string;
runtimeMapRevision: string;
}

/** Benchmark-only diagnostic payload; it is drained by the capture harness. */
export interface ModelLayerDiagnostic {
snapshot_revision: string;
runtime_map_revision: string;
snapshot_offline_count: number;
runtime_map_relevant_state: { revision: string; offline_or_not_running_service_count: number; offline_or_not_running_container_count: number };
coherent_pair_accepted: { accepted: boolean; snapshot_revision: string; runtime_map_revision: string };
derived_model_offline_value: number;
story_offline_value_pre_render: number;
rendered_home_offline_value: number | null;
fixture_generation: number | null;
monotonic_timestamp: number;
}

declare global {
interface Window {
/** Benchmark build only. Absent from the production bundle. */
__dockermapBenchAcceptanceSink?: ModelAcceptanceEvent[];
/** Benchmark build only: artificial presentation delay in ms (0/absent = off). */
__dockermapBenchRenderDelayMs?: number;
/** Revision-targeted delay used by the ordinary benchmark capture. */
__dockermapBenchRenderDelayTarget?: string;
/** One-shot seam-isolation delay, armed before the next coherent acceptance. */
__dockermapBenchDelayAfterNextAcceptance?: boolean;
/** Benchmark-only timestamp at which the post-acceptance delay timer began. */
__dockermapBenchDelayStartedAt?: number;
/** Benchmark build only. Drained synchronously by the capture harness. */
__dockermapBenchLayerSink?: ModelLayerDiagnostic[];
/** Benchmark build only. Set by the harness before it advances a fixture. */
__dockermapBenchFixtureGeneration?: number;
}
}

const sink: ModelAcceptanceEvent[] = [];
const layerSink: ModelLayerDiagnostic[] = [];
let sequence = 0;
let lastAcceptedPair: string | null = null;

/**
* The acceptance seam. Called from the real model publication path in
* `useSystemModel` at the moment `buildModel()` output becomes the model the UI
* uses — NOT from a DOM mutation, and NOT from a copied benchmark implementation.
*
* Duplicate calls for the same revision (a re-render recomputing the memo) are
* ignored, so one accepted revision produces exactly one event.
*/
export function recordModelAcceptance(snapshotRevision: string | null, runtimeMapRevision: string | null): void {
if (!__DOCKERMAP_BENCH_ACCEPTANCE__) return;
if (!snapshotRevision || snapshotRevision !== runtimeMapRevision) return;
const pair = `${snapshotRevision}\u0000${runtimeMapRevision}`;
if (pair === lastAcceptedPair) return;
lastAcceptedPair = pair;
sequence += 1;
sink.push({ seq: sequence, at: performance.now(), revision: snapshotRevision, snapshotRevision, runtimeMapRevision });
// Bounded: the sink keeps only a recent window on a long-lived page.
if (sink.length > 256) sink.splice(0, 128);
window.__dockermapBenchAcceptanceSink = sink;
}

/** Records the actual inputs and model value at the coherent-publication seam. */
export function recordModelLayers(snapshot: DockerSnapshot, runtimeMap: RuntimeMap, model: SystemModel): void {
if (!__DOCKERMAP_BENCH_ACCEPTANCE__) return;
const runtimeStates = runtimeMap.nodes.filter((node) => /offline|stopped|dead|down|exited|not.running/i.test(`${node.status ?? ""} ${node.service?.status ?? ""}`));
const diagnostic: ModelLayerDiagnostic = {
snapshot_revision: snapshot.modelRevision,
runtime_map_revision: runtimeMap.modelRevision,
snapshot_offline_count: snapshot.containers.filter((container) => /offline|stopped|dead|down|exited|not.running/i.test(container.status)).length,
runtime_map_relevant_state: {
revision: runtimeMap.modelRevision,
offline_or_not_running_service_count: runtimeStates.filter((node) => node.service !== undefined && node.service !== null).length,
offline_or_not_running_container_count: runtimeStates.filter((node) => node.type === "container").length
},
coherent_pair_accepted: { accepted: snapshot.modelRevision === runtimeMap.modelRevision, snapshot_revision: snapshot.modelRevision, runtime_map_revision: runtimeMap.modelRevision },
derived_model_offline_value: summarize(model).offline,
story_offline_value_pre_render: summarize(model).offline,
rendered_home_offline_value: null,
fixture_generation: window.__dockermapBenchFixtureGeneration ?? null,
monotonic_timestamp: performance.now()
};
layerSink.push(diagnostic);
if (layerSink.length > 256) layerSink.splice(0, 128);
window.__dockermapBenchLayerSink = layerSink;
}

/**
* The publication seam. In the product build this is the identity function: no
* state, no effects, no observable difference. In the benchmark build it is the
* point where the artificial presentation delay control withholds a newly
* accepted publication from the render tree.
*
* The delay is keyed on the accepted revision, never on the surrounding
* publication object (which is recreated on every render) — keying on the object
* would reschedule the timer on every render instead of once per publication.
*/
export function useDeliveredModel<T>(value: T, revision: string | null): T {
if (!__DOCKERMAP_BENCH_ACCEPTANCE__) return value;
return useDelayedPublication(value, revision, window.__dockermapBenchRenderDelayTarget ?? null, window.__dockermapBenchDelayAfterNextAcceptance === true);
}

/**
* Benchmark-only: stamps the accepted revision token on the document root in the
* same commit that renders the accepted model, so the capture can attribute a
* DOM repaint to a revision instead of assuming it.
*/
export function ModelAcceptanceStamp({ revision }: { revision: string | null }): ReactElement | null {
if (!__DOCKERMAP_BENCH_ACCEPTANCE__) return null;
return <AcceptedRevisionStamp revision={revision} />;
}

function AcceptedRevisionStamp({ revision }: { revision: string | null }): null {
useLayoutEffect(() => {
const root = document.documentElement;
if (revision && revision.length > 0) root.dataset.dockermapAcceptedRevision = revision;
else delete root.dataset.dockermapAcceptedRevision;
// This runs in the commit containing Home. Read its displayed metric instead
// of deriving a value from the revision or fixture generation.
const metric = [...document.querySelectorAll(".metric")].find((element) =>
element.querySelector(".metric-label")?.textContent?.trim() === "Offline"
);
const displayed = metric?.querySelector(".metric-value")?.textContent?.trim();
const latest = layerSink[layerSink.length - 1];
if (latest && latest.snapshot_revision === revision && displayed !== undefined && displayed !== "") {
const parsed = Number(displayed);
latest.rendered_home_offline_value = Number.isFinite(parsed) ? parsed : null;
}
}, [revision]);
return null;
}

function useDelayedPublication<T>(value: T, revision: string | null, delayTarget: string | null, delayAfterNextAcceptance: boolean): T {
const deliveredRevision = useRef(revision);
const isNextAcceptance = delayAfterNextAcceptance && Boolean(revision) && revision !== deliveredRevision.current;
const delayMs = revision === delayTarget || isNextAcceptance ? armedDelayMs() : 0;
const latest = useRef(value);
latest.current = value;
const [delivered, setDelivered] = useState(value);
useEffect(() => {
if (delayMs <= 0) return undefined;
// This flag is consumed only after recordModelAcceptance() ran in the same
// render. It deliberately identifies no daemon publication or trigger.
if (isNextAcceptance) window.__dockermapBenchDelayAfterNextAcceptance = false;
if (isNextAcceptance) window.__dockermapBenchDelayStartedAt = performance.now();
const timer = window.setTimeout(() => {
deliveredRevision.current = revision;
setDelivered(latest.current);
}, delayMs);
return () => window.clearTimeout(timer);
}, [revision, delayMs, isNextAcceptance]);
return delayMs > 0 ? delivered : value;
}

function armedDelayMs(): number {
if (typeof window === "undefined") return 0;
const raw = Number(window.__dockermapBenchRenderDelayMs ?? 0);
return Number.isFinite(raw) && raw > 0 ? raw : 0;
}
Loading
Loading