From a235adcdc5f0458c9601a8ed25557a42fa64b785 Mon Sep 17 00:00:00 2001 From: sam Date: Wed, 23 Sep 2026 20:19:17 +0800 Subject: [PATCH 1/4] Show active sandbox count and observed memory --- apps/web/e2e/agents-lifecycle.spec.ts | 9 +-- .../features/dashboard/DashboardView.test.tsx | 1 + .../dashboard/RuntimeTrendCharts.test.tsx | 11 ++-- .../features/dashboard/RuntimeTrendCharts.tsx | 28 +++++----- .../dashboard/runtime-history.test.ts | 55 +++++++++++++++++-- .../src/features/dashboard/runtime-history.ts | 31 ++++++++--- .../features/dashboard/runtime-trends.test.ts | 24 ++++++++ .../src/features/dashboard/runtime-trends.ts | 16 +++++- .../runtime-observability-design.md | 17 ++++-- 9 files changed, 151 insertions(+), 41 deletions(-) diff --git a/apps/web/e2e/agents-lifecycle.spec.ts b/apps/web/e2e/agents-lifecycle.spec.ts index 2dd5df09a..7b4b5626b 100644 --- a/apps/web/e2e/agents-lifecycle.spec.ts +++ b/apps/web/e2e/agents-lifecycle.spec.ts @@ -2680,7 +2680,7 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand await expect(dashboard.locator(".dashboard-runtime-sample-count")).toContainText("1 sample ·"); await expect(dashboard.getByRole("heading", { name: "CPU usage" })).toBeVisible(); await expect(dashboard.getByRole("heading", { name: "Memory usage" })).toBeVisible(); - await expect(dashboard.getByRole("heading", { name: "Compute uptime" })).toBeVisible(); + await expect(dashboard.getByRole("heading", { name: "Active Sandboxes" })).toBeVisible(); await expect(dashboard.getByRole("heading", { name: "Token throughput" })).toBeVisible(); await expect(dashboard.getByLabel("Live Runtime sampling every 30 seconds")).toBeVisible(); const liveRange = dashboard.getByRole("group", { name: "Runtime live range" }); @@ -2692,7 +2692,7 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand await refresh.click(); await expect(dashboard.getByLabel("CPU usage: 3 live samples")).toBeVisible(); await expect(dashboard.getByLabel("Memory usage: 3 live samples")).toBeVisible(); - await expect(dashboard.getByLabel("Compute uptime: 3 live samples")).toBeVisible(); + await expect(dashboard.getByLabel("Active Sandboxes: 3 live samples")).toBeVisible(); await expect(dashboard.getByLabel("Token throughput: 3 live samples")).toBeVisible(); await expect(dashboard.locator(".dashboard-runtime-sample-count")).toContainText("3 samples"); await expect(dashboard.getByText("CPU usage live trend available")).toBeAttached(); @@ -2760,7 +2760,7 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand const keyboardSelectedAt = Number(await cpuChart.getAttribute("data-selected-at")); expect(keyboardSelectedAt).toBeGreaterThanOrEqual(zoomedViewStart); expect(keyboardSelectedAt).toBeLessThanOrEqual(zoomedViewEnd); - for (const chartName of ["Memory usage", "Compute uptime", "Token throughput"]) { + for (const chartName of ["Memory usage", "Active Sandboxes", "Token throughput"]) { await expect(dashboard.getByLabel(`${chartName}: 3 live samples`)).toHaveAttribute("data-view-start", String(initialViewStart)); await expect(dashboard.getByLabel(`${chartName}: 3 live samples`)).toHaveAttribute("data-view-end", String(initialViewEnd)); } @@ -2910,10 +2910,11 @@ test("restores retained Runtime history after a Dashboard reload", async ({ page await expect(dashboard.getByLabel(/Durable · 30s; 1 Runtime targets/)).toBeVisible(); await expect(dashboard.getByLabel("Runtime durable-history charts")).toBeVisible(); await expect(dashboard.getByRole("heading", { name: "Compute uptime", exact: true })).toHaveCount(0); - await expect(dashboard.locator('[data-chart-engine="uplot"]')).toHaveCount(3); + await expect(dashboard.locator('[data-chart-engine="uplot"]')).toHaveCount(4); await expect(dashboard.getByText("CPU usage durable trend available")).toBeAttached(); await expect(dashboard).toContainText("120 buckets"); await expect(dashboard).toContainText("119/120 observations"); + await expect(dashboard.getByText("Active Sandboxes durable trend available")).toBeAttached(); await expect(dashboard.getByText("Token throughput durable trend available")).toBeAttached(); const durableCpuChart = dashboard.getByLabel("CPU usage: 120 retained buckets"); await expect(dashboard.getByRole("region", { name: "CPU usage durable history chart" })).toBeVisible(); diff --git a/apps/web/src/features/dashboard/DashboardView.test.tsx b/apps/web/src/features/dashboard/DashboardView.test.tsx index 3cb3e67a7..a3ec29127 100644 --- a/apps/web/src/features/dashboard/DashboardView.test.tsx +++ b/apps/web/src/features/dashboard/DashboardView.test.tsx @@ -257,6 +257,7 @@ describe("Dashboard loaded-result presentation", () => { expect(html).toContain("CPU usage"); expect(html).toContain("Memory usage"); expect(html).not.toContain("Compute uptime"); + expect(html).toContain("Active Sandboxes"); expect(html).toContain("Token throughput"); expect(html).toContain("No retained CPU samples"); expect(html).toContain("0/2 valid points · 0 snapshots · no history is synthesized"); diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx index c29e9be82..80052f604 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx @@ -7,6 +7,7 @@ import type { RuntimeTrendSample } from "./runtime-trends"; function sample(sampledAt: number, cpuRatio: number | null): RuntimeTrendSample { return { sampledAt, + activeSandboxCount: 1, targets: [{ seriesId: "session-1:allocation-1", label: "Runtime worker", @@ -38,7 +39,7 @@ describe("Runtime live-window chart accessibility", () => { allSeriesHidden: true, validPoints: 0, sampleCount: 24, - emptyMessage: "No complete retained memory samples", + emptyMessage: "No retained observed memory samples", })).toBe("Memory usage all series hidden; use the legend to show a series"); }); @@ -58,7 +59,7 @@ describe("Runtime live-window chart accessibility", () => { expect(html.match(/data-chart-engine="uplot"/g)).toHaveLength(4); expect(html).not.toContain("Collecting live samples"); - expect(html).toContain("Compute uptime"); + expect(html).toContain("Active Sandboxes"); }); it("exposes interactive series, point selection, and Grafana-style in-plot range selection", () => { @@ -82,9 +83,11 @@ describe("Runtime live-window chart accessibility", () => { ); expect(html).toContain('aria-label="CPU usage durable history chart"'); - expect(html.match(/data-chart-engine="uplot"/g)).toHaveLength(3); + expect(html.match(/data-chart-engine="uplot"/g)).toHaveLength(4); expect(html).not.toContain("Compute uptime"); expect(html).toContain('aria-label="CPU usage: 2 retained buckets"'); + expect(html).toContain('aria-label="Active Sandboxes durable history chart"'); + expect(html).toContain("active10"); expect(html).not.toContain('aria-label="CPU usage: 2 live samples"'); }); @@ -94,6 +97,6 @@ describe("Runtime live-window chart accessibility", () => { ); expect(html).toContain("Memory usage durable trend has 1 sparse valid point; a line requires consecutive buckets"); - expect(html).not.toContain("Memory usage No complete retained memory samples"); + expect(html).not.toContain("Memory usage No retained observed memory samples"); }); }); diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx index 2edf56ca3..521e4e18a 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx @@ -9,7 +9,7 @@ import { import uPlot from "uplot"; import "uplot/dist/uPlot.min.css"; -import { formatDashboardBytes, formatDashboardDuration, formatDashboardTokens } from "./dashboard-model"; +import { formatDashboardBytes, formatDashboardTokens } from "./dashboard-model"; import { tokenThroughput, type RuntimeTrendSample } from "./runtime-trends"; interface TrendPoint { @@ -481,7 +481,7 @@ function TrendChart({ ); } -function targetIds(samples: readonly RuntimeTrendSample[], field: "cpuRatio" | "uptimeSeconds"): string[] { +function targetIds(samples: readonly RuntimeTrendSample[], field: "cpuRatio"): string[] { const latest = new Map(); for (const sample of samples) { for (const target of sample.targets) { @@ -515,7 +515,6 @@ export function RuntimeTrendCharts({ }) { const charts = useMemo(() => { const cpuIds = targetIds(samples, "cpuRatio"); - const uptimeIds = targetIds(samples, "uptimeSeconds"); const cpu = cpuIds.map((id, index): TrendSeries => ({ id, label: targetLabel(samples, id), @@ -524,12 +523,12 @@ export function RuntimeTrendCharts({ })); const memoryUsed = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryUsageBytes })); const memoryLimit = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryLimitBytes })); - const uptime = uptimeIds.map((id, index): TrendSeries => ({ - id, - label: targetLabel(samples, id), - tone: tones[(index + 2) % tones.length] ?? "blue", - points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.targets.find((target) => target.seriesId === id)?.uptimeSeconds ?? null })), - })); + const active = [{ + id: "active", + label: "active", + tone: "green", + points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.activeSandboxCount })), + }] satisfies TrendSeries[]; const throughput = tokenThroughput(samples); return { cpu, @@ -537,7 +536,7 @@ export function RuntimeTrendCharts({ { id: "used", label: "used", tone: "purple", points: memoryUsed }, { id: "limit", label: "configured limit", tone: "green", points: memoryLimit }, ] satisfies TrendSeries[], - uptime, + active, tokens: [ { id: "input", label: "input", tone: "orange", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.inputPerMinute })) }, { id: "output", label: "output", tone: "green", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.outputPerMinute })) }, @@ -546,7 +545,10 @@ export function RuntimeTrendCharts({ }, [samples]); const cpuMaximum = Math.max(100, ...finite(charts.cpu.flatMap((series) => series.points.map((point) => point.value)))); const memoryMaximum = Math.max(1, ...finite(charts.memory.flatMap((series) => series.points.map((point) => point.value)))); - const uptimeMaximum = Math.max(1, ...finite(charts.uptime.flatMap((series) => series.points.map((point) => point.value)))); + const activeMaximum = Math.max(1, ...finite(charts.active.flatMap((series) => series.points.map((point) => point.value)))); + const activeTicks = activeMaximum <= 4 + ? Array.from({ length: activeMaximum + 1 }, (_, index) => (activeMaximum - index) / activeMaximum) + : [1, .66, .33, 0]; const tokenMaximum = Math.max(1, ...finite(charts.tokens.flatMap((series) => series.points.map((point) => point.value)))); const newest = rangeEnd ?? samples.at(-1)?.sampledAt ?? Date.now(); const oldest = rangeStart ?? samples[0]?.sampledAt ?? newest - 60 * 60 * 1_000; @@ -555,8 +557,8 @@ export function RuntimeTrendCharts({ return (
`${Math.round(value)}%`} rangeStart={oldest} rangeEnd={newest} source={source} bands={[{ from: 0, to: 30, tone: "safe" }, { from: 30, to: 70, tone: "warning" }, { from: 70, to: 100, tone: "danger" }]} ticks={[1, .7, .3, 0]} emptyMessage={durable ? "No retained CPU samples" : undefined} /> - formatDashboardBytes(Math.round(value))} rangeStart={oldest} rangeEnd={newest} source={source} emptyMessage={durable ? "No complete retained memory samples" : undefined} /> - {!durable ? formatDashboardDuration(value)} rangeStart={oldest} rangeEnd={newest} source={source} /> : null} + formatDashboardBytes(Math.round(value))} rangeStart={oldest} rangeEnd={newest} source={source} emptyMessage={durable ? "No retained observed memory samples" : undefined} /> + `${Math.round(value)}`} rangeStart={oldest} rangeEnd={newest} source={source} ticks={activeTicks} emptyMessage={durable ? "No confirmed active Sandbox samples" : undefined} /> `${formatDashboardTokens(Math.round(value))}/min`} rangeStart={oldest} rangeEnd={newest} source={source} emptyMessage={durable ? "No retained token samples" : undefined} />
); diff --git a/apps/web/src/features/dashboard/runtime-history.test.ts b/apps/web/src/features/dashboard/runtime-history.test.ts index 58cbfcd3c..d60ede3d1 100644 --- a/apps/web/src/features/dashboard/runtime-history.test.ts +++ b/apps/web/src/features/dashboard/runtime-history.test.ts @@ -125,6 +125,7 @@ describe("Runtime Durable Dashboard history", () => { expect(samples).toHaveLength(2); expect(samples[0]).toMatchObject({ sampledAt: 130_000, + activeSandboxCount: 1, memoryUsageBytes: 512, memoryLimitBytes: 1_024, inputTokensPerMinute: null, @@ -135,6 +136,49 @@ describe("Runtime Durable Dashboard history", () => { expect(samples[1]).toMatchObject({ inputTokensPerMinute: 60, outputTokensPerMinute: 20 }); }); + it("counts and aggregates distinct Runtime allocations within one Session", () => { + const source = history(); + const secondSeries = { + ...source.series[0]!, + allocation_id: "55555555-5555-4555-8555-555555555555", + points: source.series[0]!.points.map((point) => ({ + ...point, + memory: point.memory ? { ...point.memory, usage_bytes: 128, limit_bytes: 256 } : null, + })), + }; + const samples = runtimeDurableTrendSamples([session], [{ + ...source, + series: [source.series[0]!, secondSeries], + }]); + + expect(samples[0]).toMatchObject({ + activeSandboxCount: 2, + memoryUsageBytes: 640, + memoryLimitBytes: 1_280, + }); + }); + + it("deduplicates one Runtime allocation repeated across Session histories", () => { + const second = { ...session, id: "44444444-4444-4444-8444-444444444444" } as AgentSession; + const repeated = history({ + session_id: second.id, + series: [{ + ...history().series[0]!, + points: history().series[0]!.points.map((point) => ({ + ...point, + memory: point.memory ? { ...point.memory, usage_bytes: 128, limit_bytes: 256 } : null, + })), + }], + }); + const samples = runtimeDurableTrendSamples([session, second], [history(), repeated]); + + expect(samples[0]).toMatchObject({ + activeSandboxCount: 1, + memoryUsageBytes: 128, + memoryLimitBytes: 256, + }); + }); + it("does not derive compute uptime from retained allocation starts or unavailable observations", () => { const source = history(); source.series[0]!.points[1] = { @@ -142,9 +186,10 @@ describe("Runtime Durable Dashboard history", () => { }; const samples = runtimeDurableTrendSamples([session], [source]); expect(samples.flatMap((sample) => sample.targets.map((target) => target.uptimeSeconds))).toEqual([null, null]); + expect(samples[1]?.activeSandboxCount).toBe(0); }); - it("keeps aggregate memory absent when any queried target has no memory value", () => { + it("aggregates observed memory without letting an unavailable target erase it", () => { const second = { ...session, id: "44444444-4444-4444-8444-444444444444" } as AgentSession; const secondHistory = history({ session_id: second.id, @@ -152,7 +197,8 @@ describe("Runtime Durable Dashboard history", () => { series: [], }); const samples = runtimeDurableTrendSamples([session, second], [history(), secondHistory]); - expect(samples.every((sample) => sample.memoryUsageBytes === null && sample.memoryLimitBytes === null)).toBe(true); + expect(samples.every((sample) => sample.memoryUsageBytes !== null && sample.memoryLimitBytes !== null)).toBe(true); + expect(samples[0]).toMatchObject({ memoryUsageBytes: 512, memoryLimitBytes: 1_024, activeSandboxCount: 1 }); }); it("keeps omitted buckets between distant observations as gaps", () => { @@ -189,7 +235,7 @@ describe("Runtime Durable Dashboard history", () => { expect(samples[10]?.outputTokensPerMinute).toBeNull(); for (const sample of samples.slice(1, -1)) { expect(sample).toMatchObject({ - targets: [], memoryUsageBytes: null, memoryLimitBytes: null, + activeSandboxCount: null, targets: [], memoryUsageBytes: null, memoryLimitBytes: null, inputTokensPerMinute: null, outputTokensPerMinute: null, }); } @@ -203,6 +249,7 @@ describe("Runtime Durable Dashboard history", () => { expect(samples.map((sample) => sample.sampledAt)).toEqual([100_000, 130_000, 160_000, 190_000, 205_000]); expect(samples.map((sample) => sample.memoryUsageBytes)).toEqual([null, 512, 768, null, null]); expect(samples.map((sample) => sample.targets.length)).toEqual([0, 1, 1, 0, 0]); + expect(samples.map((sample) => sample.activeSandboxCount)).toEqual([null, 1, 1, null, null]); }); it("represents an entirely missing range without fabricating zero measurements", () => { @@ -219,7 +266,7 @@ describe("Runtime Durable Dashboard history", () => { expect(samples.map((sample) => sample.sampledAt)).toEqual([130_000, 160_000, 175_000]); for (const sample of samples) { expect(sample).toMatchObject({ - targets: [], memoryUsageBytes: null, memoryLimitBytes: null, + activeSandboxCount: null, targets: [], memoryUsageBytes: null, memoryLimitBytes: null, inputTokensPerMinute: null, outputTokensPerMinute: null, }); } diff --git a/apps/web/src/features/dashboard/runtime-history.ts b/apps/web/src/features/dashboard/runtime-history.ts index 9200bf92a..093f472c0 100644 --- a/apps/web/src/features/dashboard/runtime-history.ts +++ b/apps/web/src/features/dashboard/runtime-history.ts @@ -80,7 +80,9 @@ async function mapBounded( interface MutableBucket { sampledAt: number; + hasObservationCoverage: boolean; targets: Map; + activeSandboxes: Map; memory: Map; tokens: Map; } @@ -94,7 +96,7 @@ export function runtimeDurableTrendSamples( const bucket = (sampledAt: number): MutableBucket => { let value = buckets.get(sampledAt); if (!value) { - value = { sampledAt, targets: new Map(), memory: new Map(), tokens: new Map() }; + value = { sampledAt, hasObservationCoverage: false, targets: new Map(), activeSandboxes: new Map(), memory: new Map(), tokens: new Map() }; buckets.set(sampledAt, value); } return value; @@ -105,7 +107,10 @@ export function runtimeDurableTrendSamples( for (let bucketStart = start; bucketStart < end; bucketStart += history.resolution_seconds) { bucket(Math.min(bucketStart + history.resolution_seconds, end) * 1_000); } - for (const coverage of history.coverage.buckets) bucket(coverage.end * 1_000); + for (const coverage of history.coverage.buckets) { + const value = bucket(coverage.end * 1_000); + value.hasObservationCoverage ||= coverage.observation_count > 0; + } for (const usage of history.token_usage) { bucket(usage.end * 1_000).tokens.set(history.session_id, { sampledAt: usage.sampled_at * 1_000, @@ -118,6 +123,7 @@ export function runtimeDurableTrendSamples( const label = titles.get(history.session_id) ?? "Runtime"; for (const point of series.points) { const value = bucket(point.end * 1_000); + value.hasObservationCoverage ||= point.observation_count > 0; const observedAt = point.last_observed_at; value.targets.set(targetID, { seriesId: targetID, @@ -125,12 +131,18 @@ export function runtimeDurableTrendSamples( cpuRatio: point.cpu?.utilization_ratio ?? null, uptimeSeconds: null, }); + if (observedAt !== null && point.observed_count > 0) { + const previous = value.activeSandboxes.get(series.allocation_id); + if (!previous || observedAt >= previous.observedAt) { + value.activeSandboxes.set(series.allocation_id, { observedAt }); + } + } const usage = point.memory?.usage_bytes; const limit = point.memory?.limit_bytes; if (observedAt !== null && usage != null && limit != null) { - const previous = value.memory.get(history.session_id); + const previous = value.memory.get(series.allocation_id); if (!previous || observedAt >= previous.observedAt) { - value.memory.set(history.session_id, { observedAt, usage, limit }); + value.memory.set(series.allocation_id, { observedAt, usage, limit }); } } } @@ -138,16 +150,17 @@ export function runtimeDurableTrendSamples( } const samples = [...buckets.values()].sort((left, right) => left.sampledAt - right.sampledAt).map((value) => { - const completeMemory = sessions.length > 0 && value.memory.size === sessions.length; + const observedMemory = [...value.memory.values()]; return { sampledAt: value.sampledAt, + activeSandboxCount: value.hasObservationCoverage ? value.activeSandboxes.size : null, targets: [...value.targets.values()], cpuCandidates: [], - memoryUsageBytes: completeMemory - ? [...value.memory.values()].reduce((total, current) => total + current.usage, 0) + memoryUsageBytes: observedMemory.length > 0 + ? observedMemory.reduce((total, current) => total + current.usage, 0) : null, - memoryLimitBytes: completeMemory - ? [...value.memory.values()].reduce((total, current) => total + current.limit, 0) + memoryLimitBytes: observedMemory.length > 0 + ? observedMemory.reduce((total, current) => total + current.limit, 0) : null, tokenTotals: value.tokens.size === sessions.length ? [...value.tokens.entries()].map(([sessionId, usage]) => ({ sessionId, ...usage })) diff --git a/apps/web/src/features/dashboard/runtime-trends.test.ts b/apps/web/src/features/dashboard/runtime-trends.test.ts index 4554c7364..1587e6186 100644 --- a/apps/web/src/features/dashboard/runtime-trends.test.ts +++ b/apps/web/src/features/dashboard/runtime-trends.test.ts @@ -92,6 +92,7 @@ describe("Runtime live-window trends", () => { const sample = runtimeTrendSample(snapshot(120_000)); expect(sample).toMatchObject({ sampledAt: 120_000, + activeSandboxCount: 1, tokenTotals: [{ sessionId: "11111111-1111-4111-8111-111111111111", inputTokens: 100, @@ -107,6 +108,29 @@ describe("Runtime live-window trends", () => { })]); }); + it("deduplicates live aggregate count and memory by Runtime allocation identity", () => { + const duplicate = snapshot(120_000); + const secondSession = { + ...duplicate.sessions[0]!, + id: "44444444-4444-4444-8444-444444444444", + } as AgentSession; + const secondObservation = { + ...duplicate.observations[0]!, + id: secondSession.id, + session_id: secondSession.id, + observed_at: (duplicate.observations[0]!.observed_at ?? 0) + 1, + memory: { usage_bytes: 128, limit_bytes: 256 }, + } as RuntimeObservation; + duplicate.sessions.push(secondSession); + duplicate.observations.push(secondObservation); + + expect(runtimeTrendSample(duplicate)).toMatchObject({ + activeSandboxCount: 1, + memoryUsageBytes: 128, + memoryLimitBytes: 256, + }); + }); + it("deduplicates refreshes and bounds the rolling window", () => { let samples = appendRuntimeTrendSample([], snapshot(60_000), 120_000, 2); samples = appendRuntimeTrendSample(samples, snapshot(120_000)); diff --git a/apps/web/src/features/dashboard/runtime-trends.ts b/apps/web/src/features/dashboard/runtime-trends.ts index 8f9c9ce85..ba675c341 100644 --- a/apps/web/src/features/dashboard/runtime-trends.ts +++ b/apps/web/src/features/dashboard/runtime-trends.ts @@ -30,6 +30,7 @@ export interface RuntimeTrendCPUCandidate extends RuntimeTrendTarget { export interface RuntimeTrendSample { sampledAt: number; + activeSandboxCount: number | null; targets: RuntimeTrendTarget[]; cpuCandidates: RuntimeTrendCPUCandidate[]; memoryUsageBytes: number | null; @@ -110,9 +111,11 @@ export function runtimeTrendSample(snapshot: RuntimeDashboardSnapshot): RuntimeT const session = sessions.get(observation.session_id); if (!session || observation.status !== "observed") return []; const key = allocationKey(observation); - if (key === null) return []; + const allocationId = observation.instance.allocation_id; + if (key === null || typeof allocationId !== "string" || allocationId.length === 0) return []; return [{ seriesId: `${observation.session_id}:${key}`, + allocationId, label: sessionTitle(session), cpuRatio: reportedCpuRatio(observation), observedAt: safeInteger(observation.observed_at), @@ -141,11 +144,20 @@ export function runtimeTrendSample(snapshot: RuntimeDashboardSnapshot): RuntimeT cpuRatio: target.cpuRatio, uptimeSeconds: target.uptimeSeconds, })); - const pairedMemory = observed.filter((target) => ( + const latestByAllocation = new Map(); + for (const target of observed) { + const previous = latestByAllocation.get(target.allocationId); + if (!previous || (target.observedAt ?? -1) >= (previous.observedAt ?? -1)) { + latestByAllocation.set(target.allocationId, target); + } + } + const allocations = [...latestByAllocation.values()]; + const pairedMemory = allocations.filter((target) => ( target.memoryUsageBytes !== null && target.memoryLimitBytes !== null )); return { sampledAt: snapshot.loadedAt, + activeSandboxCount: allocations.length, targets, cpuCandidates: observed.flatMap((target): RuntimeTrendCPUCandidate[] => ( target.cpuRatio !== null || ( diff --git a/contracts/agents-api/runtime-observability-design.md b/contracts/agents-api/runtime-observability-design.md index d55958231..a72a5ffe2 100644 --- a/contracts/agents-api/runtime-observability-design.md +++ b/contracts/agents-api/runtime-observability-design.md @@ -187,10 +187,11 @@ measurement or lifecycle state. | Idle duration | future durable `idle_since` | Not available in the current design. | Container restart resets compute uptime but not allocation age. Live CPU deltas -require the same known compute start as well as the same allocation. Retained -charts show CPU, memory and tokens; uptime stays in the current/Live view because -the history contract does not supply each bucket's compute start. Dashboard labels -must not collapse these values into one generic Runtime duration. +require the same known compute start as well as the same allocation. Trend charts +show CPU, memory, confirmed active Sandbox count, and tokens. Compute uptime stays +in current target details because the history contract does not supply each +bucket's compute start. Dashboard labels must not collapse these values into one +generic Runtime duration. ## 9. Collection behavior @@ -350,7 +351,9 @@ Every bucket reports explicit observation coverage and nullable CPU/memory values. CPU utilization may be derived only from ordered cumulative counters inside one fence; successive intervals are assigned to the bucket containing their right endpoint and combined by CPU-capacity time. Memory uses the final -observed value in the bucket. Empty +observed value in the bucket. Dashboard memory totals aggregate only allocations with +a complete observed usage/limit pair in that bucket; an unavailable or released +target does not erase measurements from active targets. Empty buckets remain gaps. The service rejects cross-scope rows, duplicate series, overlapping or out-of-range buckets, unsafe provider labels, invalid numeric values, and results exceeding the total point budget. @@ -377,6 +380,10 @@ acceptance. - Observed CPU usage and known configured capacity. - Observed memory usage and known limits. - Reported Session tokens, together with the reporting Session count. +- Confirmed active Sandbox count over time. Each bucket counts managed allocations + with an observed provider sample; unavailable or timed-out samples are not + presented as confirmed active. A bucket with collection coverage but no observed + allocation is zero; a bucket without collection coverage remains a gap. - Data freshness and source coverage. Aggregates include only present measurements. Each total states its denominator, From 9cadb23b25fc77823e56113fe131387d8d840caf Mon Sep 17 00:00:00 2001 From: sam Date: Wed, 23 Sep 2026 22:42:50 +0800 Subject: [PATCH 2/4] Add sandbox lifecycle metrics --- apps/web/e2e/agents-lifecycle.spec.ts | 12 ++-- .../features/dashboard/DashboardView.test.tsx | 4 +- .../dashboard/RuntimeObservabilityContent.tsx | 2 +- .../dashboard/RuntimeTrendCharts.test.tsx | 17 +++-- .../features/dashboard/RuntimeTrendCharts.tsx | 20 +++--- .../features/dashboard/RuntimeTrendPanel.tsx | 4 +- .../dashboard/dashboard-model.test.ts | 13 +++- .../src/features/dashboard/dashboard-model.ts | 39 +++++++++++- .../dashboard/runtime-history.test.ts | 24 ++++--- .../src/features/dashboard/runtime-history.ts | 12 +++- .../features/dashboard/runtime-trends.test.ts | 58 ++++++++++++----- .../src/features/dashboard/runtime-trends.ts | 62 ++++++++----------- .../src/features/sessions/SessionsView.tsx | 1 - contracts/agents-api/openapi.yaml | 10 +++ contracts/agents-api/runtime-history-api.md | 3 + .../runtime-observability-design.md | 6 +- contracts/agents-api/runtime-observability.md | 11 ++++ .../agents-api/v1/runtime_observations.go | 1 + packages/agents-client/src/client.test.ts | 3 + packages/agents-client/src/client.ts | 10 ++- packages/agents-client/src/types.ts | 5 ++ .../internal/api/runtime_observations.go | 32 ++++++++++ .../internal/api/runtime_observations_test.go | 27 +++++++- 23 files changed, 275 insertions(+), 101 deletions(-) diff --git a/apps/web/e2e/agents-lifecycle.spec.ts b/apps/web/e2e/agents-lifecycle.spec.ts index 599ef5bf2..c62b31538 100644 --- a/apps/web/e2e/agents-lifecycle.spec.ts +++ b/apps/web/e2e/agents-lifecycle.spec.ts @@ -2656,6 +2656,7 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand device_id: null, connection_generation: null, }, + lifecycle_state: "active", status: "observed", reason: null, allocation_created_at: baseline - 8_500, @@ -2680,7 +2681,7 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand await expect(dashboard.locator(".dashboard-runtime-sample-count")).toContainText("1 sample ·"); await expect(dashboard.getByRole("heading", { name: "CPU usage" })).toBeVisible(); await expect(dashboard.getByRole("heading", { name: "Memory usage" })).toBeVisible(); - await expect(dashboard.getByRole("heading", { name: "Compute uptime" })).toBeVisible(); + await expect(dashboard.getByRole("heading", { name: "Runtime active" })).toBeVisible(); await expect(dashboard.getByRole("heading", { name: "Token throughput" })).toBeVisible(); await expect(dashboard.getByLabel("Live Runtime sampling every 30 seconds")).toBeVisible(); const liveRange = dashboard.getByRole("group", { name: "Runtime live range" }); @@ -2692,7 +2693,7 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand await refresh.click(); await expect(dashboard.getByLabel("CPU usage: 3 live samples")).toBeVisible(); await expect(dashboard.getByLabel("Memory usage: 3 live samples")).toBeVisible(); - await expect(dashboard.getByLabel("Compute uptime: 3 live samples")).toBeVisible(); + await expect(dashboard.getByLabel("Runtime active: 3 live samples")).toBeVisible(); await expect(dashboard.getByLabel("Token throughput: 3 live samples")).toBeVisible(); await expect(dashboard.locator(".dashboard-runtime-sample-count")).toContainText("3 samples"); await expect(dashboard.getByText("CPU usage live trend available")).toBeAttached(); @@ -2760,7 +2761,7 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand const keyboardSelectedAt = Number(await cpuChart.getAttribute("data-selected-at")); expect(keyboardSelectedAt).toBeGreaterThanOrEqual(zoomedViewStart); expect(keyboardSelectedAt).toBeLessThanOrEqual(zoomedViewEnd); - for (const chartName of ["Memory usage", "Compute uptime", "Token throughput"]) { + for (const chartName of ["Memory usage", "Runtime active", "Token throughput"]) { await expect(dashboard.getByLabel(`${chartName}: 3 live samples`)).toHaveAttribute("data-view-start", String(initialViewStart)); await expect(dashboard.getByLabel(`${chartName}: 3 live samples`)).toHaveAttribute("data-view-end", String(initialViewEnd)); } @@ -2838,6 +2839,7 @@ test("restores retained Runtime history after a Dashboard reload", async ({ page id: sessionId, object: "agent.runtime_observation", session_id: sessionId, environment_id: environmentId, mode: "openai_hosted", provider_type: "docker", instance: { kind: "managed_allocation", allocation_id: allocationId, device_id: null, connection_generation: null }, + lifecycle_state: "active", status: "observed", reason: null, allocation_created_at: now - 600, resolved_at: now, observed_at: now - 1, started_at: now - 600, cpu: { usage_seconds_total: 120, capacity_cores: 2, usage_cores: null, utilization_ratio: null }, @@ -2909,8 +2911,8 @@ test("restores retained Runtime history after a Dashboard reload", async ({ page await expect(dashboard.getByRole("group", { name: "Runtime trend source" })).toHaveCount(0); await expect(dashboard.getByLabel(/Durable · 30s; 1 Runtime targets/)).toBeVisible(); await expect(dashboard.getByLabel("Runtime durable-history charts")).toBeVisible(); - await expect(dashboard.getByRole("heading", { name: "Compute uptime", exact: true })).toHaveCount(0); - await expect(dashboard.locator('[data-chart-engine="uplot"]')).toHaveCount(3); + await expect(dashboard.getByRole("heading", { name: "Runtime active", exact: true })).toBeVisible(); + await expect(dashboard.locator('[data-chart-engine="uplot"]')).toHaveCount(4); await expect(dashboard.getByText("CPU usage durable trend available")).toBeAttached(); await expect(dashboard).toContainText("120 buckets"); await expect(dashboard).toContainText("119/120 observations"); diff --git a/apps/web/src/features/dashboard/DashboardView.test.tsx b/apps/web/src/features/dashboard/DashboardView.test.tsx index 3cb3e67a7..329babbbb 100644 --- a/apps/web/src/features/dashboard/DashboardView.test.tsx +++ b/apps/web/src/features/dashboard/DashboardView.test.tsx @@ -228,6 +228,7 @@ describe("Dashboard loaded-result presentation", () => { device_id: null, connection_generation: null, }, + lifecycle_state: "active", status: "observed", reason: null, allocation_created_at: 1_700_000_000, @@ -256,7 +257,7 @@ describe("Dashboard loaded-result presentation", () => { expect(html).toContain('aria-pressed="true">1h'); expect(html).toContain("CPU usage"); expect(html).toContain("Memory usage"); - expect(html).not.toContain("Compute uptime"); + expect(html).toContain("Runtime active"); expect(html).toContain("Token throughput"); expect(html).toContain("No retained CPU samples"); expect(html).toContain("0/2 valid points · 0 snapshots · no history is synthesized"); @@ -343,6 +344,7 @@ describe("Dashboard loaded-result presentation", () => { mode: "openai_hosted", provider_type: "docker", instance: { kind: "managed_allocation", allocation_id: "33333333-3333-4333-8333-333333333333", device_id: null, connection_generation: null }, + lifecycle_state: "active", status: "observed", reason: null, allocation_created_at: 1_700_000_000, diff --git a/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx b/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx index f52cf5c47..29859ccab 100644 --- a/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx +++ b/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx @@ -292,7 +292,7 @@ export function RuntimeObservabilityContent({ return ( <>
- } label="Active Runtimes" value={summary.observedRuntimeCount.toLocaleString("en-US")} detail={`${summary.managedRuntimeCount} managed · ${summary.unavailableRuntimeCount} unavailable`} /> + } label="Sandbox state" value={`${summary.activeSandboxCount.toLocaleString("en-US")} active · ${summary.sleepingSandboxCount.toLocaleString("en-US")} sleeping`} detail={`${summary.sandboxTotalCount.toLocaleString("en-US")} total · ${(summary.transitioningSandboxCount + summary.pendingSandboxCount).toLocaleString("en-US")} transitioning or pending`} /> } label="Cumulative CPU / capacity" value={summary.cpuUsageSecondsTotal === null && summary.cpuCapacityCores === null ? "No current sample" : `${formatDashboardDuration(summary.cpuUsageSecondsTotal)} / ${summary.cpuCapacityCores?.toLocaleString("en-US") ?? "—"} cores`} detail={`${summary.cpuCoverageCount}/${summary.observedRuntimeCount} observed Runtimes report CPU time`} /> } label="Memory now" value={summary.memoryUsageBytes === null && summary.memoryLimitBytes === null ? "No current sample" : `${formatDashboardBytes(summary.memoryUsageBytes)} / ${formatDashboardBytes(summary.memoryLimitBytes)}`} detail={`${summary.memoryCoverageCount}/${summary.observedRuntimeCount} observed Runtimes report usage`} /> } label="Reported tokens" value={formatDashboardTokens(summary.totalTokens)} detail={`${summary.tokenCoverageCount}/${summary.sessionCount} Sessions report usage`} /> diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx index 491109507..842bf1b6e 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx @@ -11,7 +11,7 @@ function sample(sampledAt: number, cpuRatio: number | null): RuntimeTrendSample seriesId: "session-1:allocation-1", label: "Runtime worker", cpuRatio, - uptimeSeconds: 120, + runtimeActive: 1, }], cpuCandidates: [], memoryUsageBytes: 512, @@ -58,7 +58,7 @@ describe("Runtime live-window chart accessibility", () => { expect(html.match(/data-chart-engine="uplot"/g)).toHaveLength(4); expect(html).not.toContain("Collecting live samples"); - expect(html).toContain("Compute uptime"); + expect(html).toContain("Runtime active"); }); it("exposes interactive series, point selection, and Grafana-style in-plot range selection", () => { @@ -82,21 +82,20 @@ describe("Runtime live-window chart accessibility", () => { ); expect(html).toContain('aria-label="CPU usage durable history chart"'); - expect(html.match(/data-chart-engine="uplot"/g)).toHaveLength(3); - expect(html).not.toContain("Compute uptime"); + expect(html.match(/data-chart-engine="uplot"/g)).toHaveLength(4); + expect(html).toContain("Runtime active"); expect(html).toContain('aria-label="CPU usage: 2 retained buckets"'); expect(html).not.toContain('aria-label="CPU usage: 2 live samples"'); }); - it("can retain an honest uptime card when a consumer requires four metric panels", () => { + it("renders retained Runtime activity as a binary chart", () => { const html = renderToStaticMarkup( - , + , ); expect(html.match(/data-chart-engine="uplot"/g)).toHaveLength(4); - expect(html).toContain("Compute uptime"); - expect(html).toContain("Live-only metric"); - expect(html).toContain("Select Live to inspect current Runtime uptime"); + expect(html).toContain("Runtime active"); + expect(html).toContain("1 active / 0 inactive"); }); it("announces an isolated durable value as sparse rather than empty", () => { diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx index a11c299e3..d8820c7d4 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx @@ -9,7 +9,7 @@ import { import uPlot from "uplot"; import "uplot/dist/uPlot.min.css"; -import { formatDashboardBytes, formatDashboardDuration, formatDashboardTokens } from "./dashboard-model"; +import { formatDashboardBytes, formatDashboardTokens } from "./dashboard-model"; import { tokenThroughput, type RuntimeTrendSample } from "./runtime-trends"; interface TrendPoint { @@ -22,6 +22,7 @@ interface TrendSeries { label: string; tone: "orange" | "green" | "blue" | "purple"; points: TrendPoint[]; + stepped?: boolean; } interface TrendBand { @@ -253,6 +254,7 @@ function TrendChart({ stroke: toneColors[entry.tone], width: 2, spanGaps: false, + paths: entry.stepped ? uPlot.paths.stepped!({ align: 1 }) : undefined, points: { show: (plot, seriesIndex, first, last) => runtimeChartShowsSparsePoints( Array.from(plot.data[seriesIndex] ?? []).slice(first, last + 1), @@ -481,7 +483,7 @@ function TrendChart({ ); } -function targetIds(samples: readonly RuntimeTrendSample[], field: "cpuRatio" | "uptimeSeconds"): string[] { +function targetIds(samples: readonly RuntimeTrendSample[], field: "cpuRatio" | "runtimeActive"): string[] { const latest = new Map(); for (const sample of samples) { for (const target of sample.targets) { @@ -507,17 +509,15 @@ export function RuntimeTrendCharts({ source = "live", rangeStart, rangeEnd, - showDurableUptimePlaceholder = false, }: { samples: readonly RuntimeTrendSample[]; source?: RuntimeTrendSource; rangeStart?: number; rangeEnd?: number; - showDurableUptimePlaceholder?: boolean; }) { const charts = useMemo(() => { const cpuIds = targetIds(samples, "cpuRatio"); - const uptimeIds = targetIds(samples, "uptimeSeconds"); + const activeIds = targetIds(samples, "runtimeActive"); const cpu = cpuIds.map((id, index): TrendSeries => ({ id, label: targetLabel(samples, id), @@ -526,11 +526,12 @@ export function RuntimeTrendCharts({ })); const memoryUsed = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryUsageBytes })); const memoryLimit = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryLimitBytes })); - const uptime = uptimeIds.map((id, index): TrendSeries => ({ + const active = activeIds.map((id, index): TrendSeries => ({ id, label: targetLabel(samples, id), tone: tones[(index + 2) % tones.length] ?? "blue", - points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.targets.find((target) => target.seriesId === id)?.uptimeSeconds ?? null })), + stepped: true, + points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.targets.find((target) => target.seriesId === id)?.runtimeActive ?? 0 })), })); const throughput = tokenThroughput(samples); return { @@ -539,7 +540,7 @@ export function RuntimeTrendCharts({ { id: "used", label: "used", tone: "purple", points: memoryUsed }, { id: "limit", label: "configured limit", tone: "green", points: memoryLimit }, ] satisfies TrendSeries[], - uptime, + active, tokens: [ { id: "input", label: "input", tone: "orange", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.inputPerMinute })) }, { id: "output", label: "output", tone: "green", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.outputPerMinute })) }, @@ -548,7 +549,6 @@ export function RuntimeTrendCharts({ }, [samples]); const cpuMaximum = Math.max(100, ...finite(charts.cpu.flatMap((series) => series.points.map((point) => point.value)))); const memoryMaximum = Math.max(1, ...finite(charts.memory.flatMap((series) => series.points.map((point) => point.value)))); - const uptimeMaximum = Math.max(1, ...finite(charts.uptime.flatMap((series) => series.points.map((point) => point.value)))); const tokenMaximum = Math.max(1, ...finite(charts.tokens.flatMap((series) => series.points.map((point) => point.value)))); const newest = rangeEnd ?? samples.at(-1)?.sampledAt ?? Date.now(); const oldest = rangeStart ?? samples[0]?.sampledAt ?? newest - 60 * 60 * 1_000; @@ -558,7 +558,7 @@ export function RuntimeTrendCharts({
`${Math.round(value)}%`} rangeStart={oldest} rangeEnd={newest} source={source} bands={[{ from: 0, to: 30, tone: "safe" }, { from: 30, to: 70, tone: "warning" }, { from: 70, to: 100, tone: "danger" }]} ticks={[1, .7, .3, 0]} emptyMessage={durable ? "No retained CPU samples" : undefined} /> formatDashboardBytes(Math.round(value))} rangeStart={oldest} rangeEnd={newest} source={source} emptyMessage={durable ? "No complete retained memory samples" : undefined} /> - {!durable || showDurableUptimePlaceholder ? formatDashboardDuration(value)} rangeStart={oldest} rangeEnd={newest} source={source} emptyMessage={durable ? "Live-only metric" : undefined} emptyDetail={durable ? "Select Live to inspect current Runtime uptime" : undefined} /> : null} + value >= .5 ? "Active" : "Inactive"} rangeStart={oldest} rangeEnd={newest} source={source} ticks={[1, 0]} emptyMessage={durable ? "No retained Runtime activity" : undefined} /> `${formatDashboardTokens(Math.round(value))}/min`} rangeStart={oldest} rangeEnd={newest} source={source} emptyMessage={durable ? "No retained token samples" : undefined} />
); diff --git a/apps/web/src/features/dashboard/RuntimeTrendPanel.tsx b/apps/web/src/features/dashboard/RuntimeTrendPanel.tsx index a8683d4bd..215012fac 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendPanel.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendPanel.tsx @@ -29,7 +29,6 @@ export function RuntimeTrendPanel({ loadRuntimeHistory, headingId = "dashboard-runtime-live-heading", title = "Resource trends", - showDurableUptimePlaceholder = false, allowSourceSelection = false, }: { snapshot: RuntimeDashboardSnapshot; @@ -37,7 +36,6 @@ export function RuntimeTrendPanel({ loadRuntimeHistory: RuntimeHistoryLoader; headingId?: string; title?: string; - showDurableUptimePlaceholder?: boolean; allowSourceSelection?: boolean; }) { const [trendSamples, setTrendSamples] = useState(() => appendRuntimeTrendSample([], snapshot)); @@ -154,7 +152,7 @@ export function RuntimeTrendPanel({ {durableState === "failed" && durableError ?

Durable history refresh failed: {durableError}

: null} {durableState === "unavailable" ?

Durable history is not configured; Live samples remain available.

: null} - + ); } diff --git a/apps/web/src/features/dashboard/dashboard-model.test.ts b/apps/web/src/features/dashboard/dashboard-model.test.ts index 03a167088..cf213359d 100644 --- a/apps/web/src/features/dashboard/dashboard-model.test.ts +++ b/apps/web/src/features/dashboard/dashboard-model.test.ts @@ -244,6 +244,7 @@ describe("Dashboard loaded-snapshot model", () => { mode: "openai_hosted", provider_type: "docker", instance: { kind: "managed_allocation", allocation_id: "44444444-4444-4444-8444-444444444444", device_id: null, connection_generation: null }, + lifecycle_state: "active", status: "observed", reason: null, allocation_created_at: 100, @@ -260,6 +261,7 @@ describe("Dashboard loaded-snapshot model", () => { mode: "none", provider_type: null, instance: { kind: "none", allocation_id: null, device_id: null, connection_generation: null }, + lifecycle_state: null, status: "unsupported", reason: "runtime_mode_not_observable", allocation_created_at: null, @@ -274,6 +276,11 @@ describe("Dashboard loaded-snapshot model", () => { expect(model.summary).toMatchObject({ sessionCount: 2, managedRuntimeCount: 1, + sandboxTotalCount: 1, + activeSandboxCount: 1, + sleepingSandboxCount: 0, + transitioningSandboxCount: 0, + pendingSandboxCount: 0, observedRuntimeCount: 1, unavailableRuntimeCount: 0, unsupportedRuntimeCount: 1, @@ -310,6 +317,7 @@ describe("Dashboard loaded-snapshot model", () => { device_id: null, connection_generation: null, }, + lifecycle_state: "stopped", status: "unavailable", reason: "runtime_not_running", allocation_created_at: 100, @@ -320,7 +328,9 @@ describe("Dashboard loaded-snapshot model", () => { memory: null, }; - expect(buildRuntimeDashboardModel([stopped], [observation]).rows[0]?.allocationAgeSeconds).toBeNull(); + const model = buildRuntimeDashboardModel([stopped], [observation]); + expect(model.rows[0]?.allocationAgeSeconds).toBeNull(); + expect(model.summary).toMatchObject({ sandboxTotalCount: 0, activeSandboxCount: 0, sleepingSandboxCount: 0 }); }); it("does not count capacity-only or limit-only samples as usage coverage", () => { @@ -349,6 +359,7 @@ describe("Dashboard loaded-snapshot model", () => { device_id: null, connection_generation: null, }, + lifecycle_state: "active", status: "observed", reason: null, allocation_created_at: null, diff --git a/apps/web/src/features/dashboard/dashboard-model.ts b/apps/web/src/features/dashboard/dashboard-model.ts index 5c31b55f1..e43715be5 100644 --- a/apps/web/src/features/dashboard/dashboard-model.ts +++ b/apps/web/src/features/dashboard/dashboard-model.ts @@ -56,6 +56,11 @@ export interface RuntimeDashboardRow { export interface RuntimeDashboardSummary { sessionCount: number; managedRuntimeCount: number; + sandboxTotalCount: number; + activeSandboxCount: number; + sleepingSandboxCount: number; + transitioningSandboxCount: number; + pendingSandboxCount: number; observedRuntimeCount: number; unavailableRuntimeCount: number; unsupportedRuntimeCount: number; @@ -282,6 +287,11 @@ export function buildRuntimeDashboardModel( const sessionsById = new Map(sessions.map((session) => [session.id, session])); const rows: RuntimeDashboardRow[] = []; let managedRuntimeCount = 0; + let sandboxTotalCount = 0; + let activeSandboxCount = 0; + let sleepingSandboxCount = 0; + let transitioningSandboxCount = 0; + let pendingSandboxCount = 0; let observedRuntimeCount = 0; let unavailableRuntimeCount = 0; let unsupportedRuntimeCount = 0; @@ -315,7 +325,29 @@ export function buildRuntimeDashboardModel( oldestResolvedAt = oldestResolvedAt === null ? resolvedAt : Math.min(oldestResolvedAt, resolvedAt); newestResolvedAt = newestResolvedAt === null ? resolvedAt : Math.max(newestResolvedAt, resolvedAt); } - if (observation.mode === "openai_hosted") managedRuntimeCount += 1; + if (observation.mode === "openai_hosted") { + managedRuntimeCount += 1; + switch (observation.lifecycle_state) { + case "active": + activeSandboxCount += 1; + sandboxTotalCount += 1; + break; + case "sleeping": + sleepingSandboxCount += 1; + sandboxTotalCount += 1; + break; + case "transitioning": + transitioningSandboxCount += 1; + sandboxTotalCount += 1; + break; + case "pending": + pendingSandboxCount += 1; + sandboxTotalCount += 1; + break; + case "stopped": + break; + } + } if (observation.status === "observed") { observedRuntimeCount += 1; const cpuUsage = safeFiniteNonNegative(observation.cpu?.usage_seconds_total); @@ -388,6 +420,11 @@ export function buildRuntimeDashboardModel( summary: { sessionCount: rows.length, managedRuntimeCount, + sandboxTotalCount, + activeSandboxCount, + sleepingSandboxCount, + transitioningSandboxCount, + pendingSandboxCount, observedRuntimeCount, unavailableRuntimeCount, unsupportedRuntimeCount, diff --git a/apps/web/src/features/dashboard/runtime-history.test.ts b/apps/web/src/features/dashboard/runtime-history.test.ts index 58cbfcd3c..14d148e41 100644 --- a/apps/web/src/features/dashboard/runtime-history.test.ts +++ b/apps/web/src/features/dashboard/runtime-history.test.ts @@ -129,19 +129,25 @@ describe("Runtime Durable Dashboard history", () => { memoryLimitBytes: 1_024, inputTokensPerMinute: null, outputTokensPerMinute: null, - targets: [{ label: "Durable worker", cpuRatio: .25, uptimeSeconds: null }], }); - expect(samples[1]?.targets[0]?.uptimeSeconds).toBeNull(); + expect(samples[0]?.targets).toEqual(expect.arrayContaining([ + expect.objectContaining({ label: "Durable worker", cpuRatio: null, runtimeActive: 1 }), + expect.objectContaining({ label: "Durable worker", cpuRatio: .25, runtimeActive: null }), + ])); + expect(samples[1]?.targets.find((target) => target.runtimeActive !== null)?.runtimeActive).toBe(1); expect(samples[1]).toMatchObject({ inputTokensPerMinute: 60, outputTokensPerMinute: 20 }); }); - it("does not derive compute uptime from retained allocation starts or unavailable observations", () => { + it("projects unavailable retained observations as inactive", () => { const source = history(); source.series[0]!.points[1] = { ...source.series[0]!.points[1]!, observed_count: 0, unavailable_count: 1, cpu: null, memory: null, }; + source.coverage.buckets[1] = { + ...source.coverage.buckets[1]!, observed_count: 0, unavailable_count: 1, + }; const samples = runtimeDurableTrendSamples([session], [source]); - expect(samples.flatMap((sample) => sample.targets.map((target) => target.uptimeSeconds))).toEqual([null, null]); + expect(samples.flatMap((sample) => sample.targets.map((target) => target.runtimeActive)).filter((value) => value !== null)).toEqual([1, 0]); }); it("keeps aggregate memory absent when any queried target has no memory value", () => { @@ -183,13 +189,13 @@ describe("Runtime Durable Dashboard history", () => { expect(samples.map((sample) => sample.sampledAt)).toEqual( Array.from({ length: 11 }, (_, index) => (130 + index * 30) * 1_000), ); - expect(samples[0]?.targets[0]?.cpuRatio).toBe(.25); - expect(samples[10]?.targets[0]?.cpuRatio).toBe(.5); + expect(samples[0]?.targets.find((target) => target.cpuRatio !== null)?.cpuRatio).toBe(.25); + expect(samples[10]?.targets.find((target) => target.cpuRatio !== null)?.cpuRatio).toBe(.5); expect(samples[10]?.inputTokensPerMinute).toBeNull(); expect(samples[10]?.outputTokensPerMinute).toBeNull(); for (const sample of samples.slice(1, -1)) { expect(sample).toMatchObject({ - targets: [], memoryUsageBytes: null, memoryLimitBytes: null, + targets: [expect.objectContaining({ cpuRatio: null, runtimeActive: 0 })], memoryUsageBytes: null, memoryLimitBytes: null, inputTokensPerMinute: null, outputTokensPerMinute: null, }); } @@ -202,7 +208,7 @@ describe("Runtime Durable Dashboard history", () => { })]); expect(samples.map((sample) => sample.sampledAt)).toEqual([100_000, 130_000, 160_000, 190_000, 205_000]); expect(samples.map((sample) => sample.memoryUsageBytes)).toEqual([null, 512, 768, null, null]); - expect(samples.map((sample) => sample.targets.length)).toEqual([0, 1, 1, 0, 0]); + expect(samples.map((sample) => sample.targets.length)).toEqual([1, 2, 2, 1, 1]); }); it("represents an entirely missing range without fabricating zero measurements", () => { @@ -219,7 +225,7 @@ describe("Runtime Durable Dashboard history", () => { expect(samples.map((sample) => sample.sampledAt)).toEqual([130_000, 160_000, 175_000]); for (const sample of samples) { expect(sample).toMatchObject({ - targets: [], memoryUsageBytes: null, memoryLimitBytes: null, + targets: [expect.objectContaining({ cpuRatio: null, runtimeActive: 0 })], memoryUsageBytes: null, memoryLimitBytes: null, inputTokensPerMinute: null, outputTokensPerMinute: null, }); } diff --git a/apps/web/src/features/dashboard/runtime-history.ts b/apps/web/src/features/dashboard/runtime-history.ts index 9200bf92a..4799e588d 100644 --- a/apps/web/src/features/dashboard/runtime-history.ts +++ b/apps/web/src/features/dashboard/runtime-history.ts @@ -102,8 +102,16 @@ export function runtimeDurableTrendSamples( for (const history of histories) { const { start, end } = history.requested_range; + const coverageByEnd = new Map(history.coverage.buckets.map((coverage) => [coverage.end, coverage])); for (let bucketStart = start; bucketStart < end; bucketStart += history.resolution_seconds) { - bucket(Math.min(bucketStart + history.resolution_seconds, end) * 1_000); + const bucketEnd = Math.min(bucketStart + history.resolution_seconds, end); + const coverage = coverageByEnd.get(bucketEnd); + bucket(bucketEnd * 1_000).targets.set(`activity:${history.session_id}`, { + seriesId: `activity:${history.session_id}`, + label: titles.get(history.session_id) ?? "Runtime", + cpuRatio: null, + runtimeActive: coverage && coverage.observed_count > 0 ? 1 : 0, + }); } for (const coverage of history.coverage.buckets) bucket(coverage.end * 1_000); for (const usage of history.token_usage) { @@ -123,7 +131,7 @@ export function runtimeDurableTrendSamples( seriesId: targetID, label, cpuRatio: point.cpu?.utilization_ratio ?? null, - uptimeSeconds: null, + runtimeActive: null, }); const usage = point.memory?.usage_bytes; const limit = point.memory?.limit_bytes; diff --git a/apps/web/src/features/dashboard/runtime-trends.test.ts b/apps/web/src/features/dashboard/runtime-trends.test.ts index 4554c7364..69746e607 100644 --- a/apps/web/src/features/dashboard/runtime-trends.test.ts +++ b/apps/web/src/features/dashboard/runtime-trends.test.ts @@ -9,6 +9,7 @@ import { runtimeTrendRange, runtimeTrendSample, tokenThroughput, + type RuntimeTrendSample, } from "./runtime-trends"; function snapshot(at: number, options: { @@ -72,6 +73,7 @@ function snapshot(at: number, options: { device_id: null, connection_generation: null, }, + lifecycle_state: "active", allocation_created_at: observedAt - 180, resolved_at: observedAt, observed_at: observedAt, @@ -88,6 +90,8 @@ function snapshot(at: number, options: { } describe("Runtime live-window trends", () => { + const cpuTarget = (sample: RuntimeTrendSample | undefined) => sample?.targets.find((target) => target.runtimeActive === null); + it("projects only honest point-in-time and cumulative Session values", () => { const sample = runtimeTrendSample(snapshot(120_000)); expect(sample).toMatchObject({ @@ -100,11 +104,35 @@ describe("Runtime live-window trends", () => { memoryUsageBytes: 512, memoryLimitBytes: 1_024, }); - expect(sample.targets).toEqual([expect.objectContaining({ + expect(sample.targets).toEqual(expect.arrayContaining([expect.objectContaining({ + label: "Runtime worker", + cpuRatio: null, + runtimeActive: 1, + }), expect.objectContaining({ label: "Runtime worker", cpuRatio: .25, - uptimeSeconds: 120, - })]); + runtimeActive: null, + })])); + }); + + it("projects a managed Session without an allocation as inactive", () => { + const pending = snapshot(120_000); + pending.observations = [{ + ...pending.observations[0]!, + instance: { kind: "managed_allocation", allocation_id: null, device_id: null, connection_generation: null }, + lifecycle_state: "pending", + status: "unavailable", + reason: "allocation_pending", + allocation_created_at: null, + observed_at: null, + started_at: null, + cpu: null, + memory: null, + } as RuntimeObservation]; + + expect(runtimeTrendSample(pending).targets).toEqual([ + expect.objectContaining({ seriesId: `activity:${pending.sessions[0]!.id}`, runtimeActive: 0 }), + ]); }); it("deduplicates refreshes and bounds the rolling window", () => { @@ -151,7 +179,7 @@ describe("Runtime live-window trends", () => { let samples = appendRuntimeTrendSample([], cumulative(60_000, 10)); samples = appendRuntimeTrendSample(samples, cumulative(120_000, 70)); samples = appendRuntimeTrendSample(samples, cumulative(180_000, 130)); - expect(samples.map((sample) => sample.targets[0]?.cpuRatio ?? null)).toEqual([null, .5, .5]); + expect(samples.map((sample) => cpuTarget(sample)?.cpuRatio ?? null)).toEqual([null, .5, .5]); }); it("uses allocation identity for Live chart series while ignoring start-time jitter", () => { @@ -162,8 +190,8 @@ describe("Runtime live-window trends", () => { startedAt: 1, allocationId: "44444444-4444-4444-8444-444444444444", })); - expect(first.targets[0]?.seriesId).toBe(jittered.targets[0]?.seriesId); - expect(replaced.targets[0]?.seriesId).not.toBe(first.targets[0]?.seriesId); + expect(cpuTarget(first)?.seriesId).toBe(cpuTarget(jittered)?.seriesId); + expect(cpuTarget(replaced)?.seriesId).not.toBe(cpuTarget(first)?.seriesId); }); it("resets cumulative CPU on start changes, allocation changes or counter regressions", () => { @@ -173,14 +201,13 @@ describe("Runtime live-window trends", () => { const continued = appendRuntimeTrendSample(appendRuntimeTrendSample([], base), snapshot(120_000, { cpuRatio: null, cpuUsageCores: null, cpuUsageSecondsTotal: 160, startedAt: 1, })); - expect(continued.at(-1)?.targets[0]?.cpuRatio ?? null).toBeNull(); - expect(continued[1]?.targets[0]?.seriesId).toBe(continued[0]?.targets[0]?.seriesId); + expect(cpuTarget(continued.at(-1))?.cpuRatio ?? null).toBeNull(); for (const next of [ snapshot(120_000, { cpuRatio: null, cpuUsageCores: null, cpuUsageSecondsTotal: 160, startedAt: 0, allocationId: "44444444-4444-4444-8444-444444444444" }), snapshot(120_000, { cpuRatio: null, cpuUsageCores: null, cpuUsageSecondsTotal: 10, startedAt: 0 }), ]) { const samples = appendRuntimeTrendSample(appendRuntimeTrendSample([], base), next); - expect(samples.at(-1)?.targets[0]?.cpuRatio ?? null).toBeNull(); + expect(cpuTarget(samples.at(-1))?.cpuRatio ?? null).toBeNull(); } }); @@ -191,26 +218,25 @@ describe("Runtime live-window trends", () => { let samples = appendRuntimeTrendSample([], cumulative(60_000, 1, 0)); samples = appendRuntimeTrendSample(samples, cumulative(90_000, 20, 65)); samples = appendRuntimeTrendSample(samples, cumulative(120_000, 50, 65)); - expect(samples.map((sample) => sample.targets[0]?.cpuRatio ?? null)).toEqual([null, null, .5]); - expect(new Set(samples.map((sample) => sample.targets[0]?.seriesId)).size).toBe(1); + expect(samples.map((sample) => cpuTarget(sample)?.cpuRatio ?? null)).toEqual([null, null, .5]); for (const [priorStart, nextStart] of [[null, 0], [0, null], [null, null]] as const) { const missingFence = appendRuntimeTrendSample( appendRuntimeTrendSample([], cumulative(60_000, 1, priorStart)), cumulative(90_000, 20, nextStart), ); - expect(missingFence.at(-1)?.targets[0]?.cpuRatio ?? null).toBeNull(); + expect(cpuTarget(missingFence.at(-1))?.cpuRatio ?? null).toBeNull(); } }); it("keeps directly reported CPU continuous across start changes but not stale observations", () => { const base = snapshot(60_000, { cpuRatio: .25, startedAt: 0 }); const continued = appendRuntimeTrendSample(appendRuntimeTrendSample([], base), snapshot(120_000, { cpuRatio: .5, startedAt: 1 })); - expect(continued.at(-1)?.targets[0]?.cpuRatio ?? null).toBe(.5); + expect(cpuTarget(continued.at(-1))?.cpuRatio ?? null).toBe(.5); const stale = appendRuntimeTrendSample( appendRuntimeTrendSample([], base), snapshot(120_000, { cpuRatio: .5, startedAt: 0, observedAt: 60 }), ); - expect(stale.at(-1)?.targets[0]?.cpuRatio ?? null).toBeNull(); + expect(cpuTarget(stale.at(-1))?.cpuRatio ?? null).toBeNull(); }); it("rejects non-finite CPU ratios produced by finite provider inputs", () => { @@ -219,7 +245,7 @@ describe("Runtime live-window trends", () => { cpuUsageCores: Number.MAX_VALUE, cpuCapacity: Number.MIN_VALUE, })); - expect(direct.targets[0]?.cpuRatio ?? null).toBeNull(); + expect(cpuTarget(direct)?.cpuRatio ?? null).toBeNull(); const cumulative = (at: number, usage: number) => snapshot(at, { cpuRatio: null, @@ -232,7 +258,7 @@ describe("Runtime live-window trends", () => { appendRuntimeTrendSample([], cumulative(60_000, 0)), cumulative(120_000, Number.MAX_VALUE), ); - expect(samples.at(-1)?.targets[0]?.cpuRatio ?? null).toBeNull(); + expect(cpuTarget(samples.at(-1))?.cpuRatio ?? null).toBeNull(); }); it("keeps Session-set churn as a gap instead of publishing partial throughput", () => { diff --git a/apps/web/src/features/dashboard/runtime-trends.ts b/apps/web/src/features/dashboard/runtime-trends.ts index 8f9c9ce85..b6f600efe 100644 --- a/apps/web/src/features/dashboard/runtime-trends.ts +++ b/apps/web/src/features/dashboard/runtime-trends.ts @@ -16,7 +16,7 @@ export interface RuntimeTrendTarget { seriesId: string; label: string; cpuRatio: number | null; - uptimeSeconds: number | null; + runtimeActive: 0 | 1 | null; } export interface RuntimeTrendCPUCandidate extends RuntimeTrendTarget { @@ -78,22 +78,13 @@ function reportedCpuRatio(observation: RuntimeObservation): number | null { } function allocationKey(observation: RuntimeObservation): string | null { - if (observation.status !== "observed") return null; + if (observation.mode !== "openai_hosted") return null; const allocationId = observation.instance.allocation_id; return typeof allocationId === "string" && allocationId.length > 0 ? `${observation.instance.kind}:${allocationId}` : null; } -function uptimeSeconds(observation: RuntimeObservation): number | null { - if (observation.status !== "observed") return null; - const startedAt = safeInteger(observation.started_at); - const observedAt = safeInteger(observation.observed_at); - return startedAt !== null && observedAt !== null && observedAt >= startedAt - ? observedAt - startedAt - : null; -} - function tokenTotals(sessions: readonly AgentSession[], sampledAt: number): RuntimeTrendTokenTotal[] { return sessions.flatMap((session): RuntimeTrendTokenTotal[] => { const inputTokens = safeInteger(session.usage?.input_tokens); @@ -106,13 +97,13 @@ function tokenTotals(sessions: readonly AgentSession[], sampledAt: number): Runt export function runtimeTrendSample(snapshot: RuntimeDashboardSnapshot): RuntimeTrendSample { const sessions = new Map(snapshot.sessions.map((session) => [session.id, session])); - const observed = snapshot.observations.flatMap((observation) => { + const managed = snapshot.observations.flatMap((observation) => { const session = sessions.get(observation.session_id); - if (!session || observation.status !== "observed") return []; + if (!session || observation.mode !== "openai_hosted") return []; const key = allocationKey(observation); - if (key === null) return []; return [{ - seriesId: `${observation.session_id}:${key}`, + sessionId: observation.session_id, + seriesId: key === null ? null : `${observation.session_id}:${key}`, label: sessionTitle(session), cpuRatio: reportedCpuRatio(observation), observedAt: safeInteger(observation.observed_at), @@ -122,25 +113,26 @@ export function runtimeTrendSample(snapshot: RuntimeDashboardSnapshot): RuntimeT capacityCores: finiteNonNegative(observation.cpu?.capacity_cores), memoryUsageBytes: finiteNonNegative(observation.memory?.usage_bytes), memoryLimitBytes: finiteNonNegative(observation.memory?.limit_bytes), - uptimeSeconds: uptimeSeconds(observation), + runtimeActive: observation.status === "observed" ? 1 as const : 0 as const, }]; }); - const targetIds = new Set([ - ...observed.filter((target) => target.cpuRatio !== null) + const activeTargets: RuntimeTrendTarget[] = managed.slice(0, RUNTIME_TREND_SERIES_LIMIT).map((target) => ({ + seriesId: `activity:${target.sessionId}`, + label: target.label, + cpuRatio: null, + runtimeActive: target.runtimeActive, + })); + const cpuTargets: RuntimeTrendTarget[] = managed.filter((target) => target.seriesId !== null && target.cpuRatio !== null) .sort((left, right) => (right.cpuRatio ?? 0) - (left.cpuRatio ?? 0)) .slice(0, RUNTIME_TREND_SERIES_LIMIT) - .map((target) => target.seriesId), - ...observed.filter((target) => target.uptimeSeconds !== null) - .sort((left, right) => (right.uptimeSeconds ?? 0) - (left.uptimeSeconds ?? 0)) - .slice(0, RUNTIME_TREND_SERIES_LIMIT) - .map((target) => target.seriesId), - ]); - const targets = observed.filter((target) => targetIds.has(target.seriesId)).map((target) => ({ - seriesId: target.seriesId, + .map((target) => ({ + seriesId: target.seriesId!, label: target.label, cpuRatio: target.cpuRatio, - uptimeSeconds: target.uptimeSeconds, + runtimeActive: null, })); + const targets = [...activeTargets, ...cpuTargets]; + const observed = managed.filter((target) => target.runtimeActive === 1); const pairedMemory = observed.filter((target) => ( target.memoryUsageBytes !== null && target.memoryLimitBytes !== null )); @@ -148,15 +140,15 @@ export function runtimeTrendSample(snapshot: RuntimeDashboardSnapshot): RuntimeT sampledAt: snapshot.loadedAt, targets, cpuCandidates: observed.flatMap((target): RuntimeTrendCPUCandidate[] => ( - target.cpuRatio !== null || ( + target.seriesId !== null && (target.cpuRatio !== null || ( target.observedAt !== null && target.allocationKey !== null && target.usageSecondsTotal !== null && target.capacityCores !== null && target.capacityCores > 0 - ) + )) ? [{ - seriesId: target.seriesId, + seriesId: target.seriesId!, label: target.label, cpuRatio: target.cpuRatio, - uptimeSeconds: target.uptimeSeconds, + runtimeActive: null, observedAt: target.observedAt, startedAt: target.startedAt, allocationKey: target.allocationKey, @@ -210,8 +202,8 @@ function cpuRatios(previous: RuntimeTrendSample, next: RuntimeTrendSample): Map< } function applyCPURatios(sample: RuntimeTrendSample, ratios: ReadonlyMap): void { - const uptime = sample.targets.filter((target) => target.uptimeSeconds !== null) - .sort((left, right) => (right.uptimeSeconds ?? 0) - (left.uptimeSeconds ?? 0)) + const active = sample.targets.filter((target) => target.runtimeActive !== null) + .sort((left, right) => (right.runtimeActive ?? 0) - (left.runtimeActive ?? 0)) .slice(0, RUNTIME_TREND_SERIES_LIMIT) .map((target) => ({ ...target, cpuRatio: null })); const cpu = sample.cpuCandidates.flatMap((candidate): RuntimeTrendTarget[] => { @@ -220,12 +212,12 @@ function applyCPURatios(sample: RuntimeTrendSample, ratios: ReadonlyMap (right.cpuRatio ?? 0) - (left.cpuRatio ?? 0)) .slice(0, RUNTIME_TREND_SERIES_LIMIT); const selected = new Map( - uptime.map((target) => [target.seriesId, target]), + active.map((target) => [target.seriesId, target]), ); for (const target of cpu) selected.set(target.seriesId, target); sample.targets = [...selected.values()]; diff --git a/apps/web/src/features/sessions/SessionsView.tsx b/apps/web/src/features/sessions/SessionsView.tsx index 54e145024..ef3be85bc 100644 --- a/apps/web/src/features/sessions/SessionsView.tsx +++ b/apps/web/src/features/sessions/SessionsView.tsx @@ -924,7 +924,6 @@ export function SessionsView({ loadRuntimeHistory={loadRuntimeHistory} headingId="session-runtime-trends-heading" title="Session resource trends" - showDurableUptimePlaceholder allowSourceSelection /> ) : runtimeError ? ( diff --git a/contracts/agents-api/openapi.yaml b/contracts/agents-api/openapi.yaml index 914bf7b12..aca37075e 100644 --- a/contracts/agents-api/openapi.yaml +++ b/contracts/agents-api/openapi.yaml @@ -1633,6 +1633,15 @@ definitions: type: string instance: $ref: '#/definitions/v1.RuntimeInstance' + lifecycle_state: + enum: + - active + - sleeping + - transitioning + - pending + - stopped + type: string + x-nullable: true memory: allOf: - $ref: '#/definitions/v1.RuntimeMemoryObservation' @@ -1686,6 +1695,7 @@ definitions: - environment_id - id - instance + - lifecycle_state - memory - mode - object diff --git a/contracts/agents-api/runtime-history-api.md b/contracts/agents-api/runtime-history-api.md index 968a7b66e..5acda837e 100644 --- a/contracts/agents-api/runtime-history-api.md +++ b/contracts/agents-api/runtime-history-api.md @@ -172,3 +172,6 @@ or malformed data reject the entire response with a 502 client projection error. Compute uptime is available from current observations only. Retained allocation series can span compute restarts and unavailable intervals; their earliest start is not a per-bucket compute start and must not be used to draw an uptime history. +Clients may project a binary Runtime activity series from bucket coverage: +`observed_count > 0` is `1`; a missing or unavailable bucket is `0`. This does +not distinguish a sleeping Runtime from a collection failure. diff --git a/contracts/agents-api/runtime-observability-design.md b/contracts/agents-api/runtime-observability-design.md index d55958231..c474b8a79 100644 --- a/contracts/agents-api/runtime-observability-design.md +++ b/contracts/agents-api/runtime-observability-design.md @@ -183,13 +183,15 @@ measurement or lifecycle state. | --- | --- | --- | | Allocation age | allocation `created_at` to `released_at` or now | Age of Core's allocation record. | | Compute uptime | provider `started_at` to sample `observed_at` | Age of the current compute incarnation. | +| Runtime active | successful observation in the selected bucket | Binary operational signal: 1 active, 0 inactive or unavailable. | | Busy duration | Turn `started_at` to `completed_at` or now | Time model work has been active. | | Idle duration | future durable `idle_since` | Not available in the current design. | Container restart resets compute uptime but not allocation age. Live CPU deltas require the same known compute start as well as the same allocation. Retained -charts show CPU, memory and tokens; uptime stays in the current/Live view because -the history contract does not supply each bucket's compute start. Dashboard labels +charts show CPU, memory, tokens, and binary Runtime activity; uptime remains a +current observation/table value because history does not supply each bucket's +compute start. Dashboard labels must not collapse these values into one generic Runtime duration. ## 9. Collection behavior diff --git a/contracts/agents-api/runtime-observability.md b/contracts/agents-api/runtime-observability.md index 305424b7c..bc7659c0b 100644 --- a/contracts/agents-api/runtime-observability.md +++ b/contracts/agents-api/runtime-observability.md @@ -41,6 +41,12 @@ rendered or aggregated as zero. A whole observation has one of three states: `observed`, `unsupported`, or `unavailable`. Provider and permission failures are errors, not ordinary unavailability. +Managed observations also expose a provider-neutral `lifecycle_state` derived +from Core's allocation and compute lifecycle: `active`, `sleeping`, +`transitioning`, `pending`, or `stopped`. Non-managed modes return `null`. +This field is current control-plane state; it is not inferred from a failed +provider sample. + Docker reports cumulative cgroup CPU time and current cgroup memory usage. CPU and memory capacity come from the inspected container configuration. Inspect and Stats are read-only; observation must not renew, restart, create, or stop the container. @@ -73,6 +79,11 @@ This phase supplies compute uptime evidence and retains the existing durable allocation and Turn timestamps. It does not infer idle time. CPU quietness, heartbeat age, connection status, and `kept_at` are not authoritative idle state. +Web projects Runtime activity as a binary chart: a successful observation in a +time bucket is `1`; an absent or unavailable observation is `0`. This operational +availability view is intentionally not a durable classification of sleeping +versus collection failure. + Future automatic suspension requires a separate durable control model, including an activity revision and timestamps such as `idle_since` and `shutdown_requested_at`. Metrics, an in-memory cache, or a monitoring backend must diff --git a/contracts/agents-api/v1/runtime_observations.go b/contracts/agents-api/v1/runtime_observations.go index d82d8d514..56d42834f 100644 --- a/contracts/agents-api/v1/runtime_observations.go +++ b/contracts/agents-api/v1/runtime_observations.go @@ -8,6 +8,7 @@ type RuntimeObservation struct { Mode string `json:"mode" enums:"none,self_hosted,openai_hosted" binding:"required"` ProviderType *string `json:"provider_type" extensions:"x-nullable" binding:"required" pattern:"^[a-z][a-z0-9_]{0,31}$"` Instance RuntimeInstance `json:"instance" binding:"required"` + LifecycleState *string `json:"lifecycle_state" extensions:"x-nullable" binding:"required" enums:"active,sleeping,transitioning,pending,stopped"` Status string `json:"status" enums:"observed,unsupported,unavailable" binding:"required"` Reason *string `json:"reason" extensions:"x-nullable" binding:"required" enums:"runtime_mode_not_observable,allocation_pending,runtime_not_running,source_not_configured,sample_timeout,sample_unavailable"` AllocationCreatedAt *int64 `json:"allocation_created_at" extensions:"x-nullable" binding:"required" minimum:"0"` diff --git a/packages/agents-client/src/client.test.ts b/packages/agents-client/src/client.test.ts index 46ac55a5f..71c43deaa 100644 --- a/packages/agents-client/src/client.test.ts +++ b/packages/agents-client/src/client.test.ts @@ -122,6 +122,7 @@ function runtimeObservation(overrides: Record = {}): Record { mode: "none", provider_type: null, instance: { kind: "none", allocation_id: null, device_id: null, connection_generation: null }, + lifecycle_state: null, status: "unsupported", reason: "runtime_mode_not_observable", allocation_created_at: null, @@ -2685,6 +2687,7 @@ describe("OpenAIAgentsClient", () => { ["unknown field", () => ({ ...runtimeObservation(), provider_native_id: "hidden" })], ["foreign Session", () => ({ ...runtimeObservation(), session_id: "55555555-5555-4555-8555-555555555555" })], ["invalid status/reason", () => ({ ...runtimeObservation(), status: "observed", reason: "sample_timeout" })], + ["invalid lifecycle state", () => ({ ...runtimeObservation(), lifecycle_state: "paused" })], ["invalid mode/instance", () => ({ ...runtimeObservation(), mode: "none" })], ["negative CPU", () => ({ ...runtimeObservation(), cpu: { usage_seconds_total: -1, capacity_cores: 2, usage_cores: null, utilization_ratio: null, diff --git a/packages/agents-client/src/client.ts b/packages/agents-client/src/client.ts index d57e3b491..36e4dd892 100644 --- a/packages/agents-client/src/client.ts +++ b/packages/agents-client/src/client.ts @@ -326,7 +326,7 @@ const unsafeUnknownEventFields = new Set([ ]); const runtimeObservationFields = new Set([ "id", "object", "session_id", "environment_id", "mode", "provider_type", "instance", "status", "reason", - "allocation_created_at", "resolved_at", "observed_at", "started_at", "cpu", "memory", + "lifecycle_state", "allocation_created_at", "resolved_at", "observed_at", "started_at", "cpu", "memory", ]); const runtimeInstanceFields = new Set(["kind", "allocation_id", "device_id", "connection_generation"]); const runtimeCPUFields = new Set(["usage_seconds_total", "capacity_cores", "usage_cores", "utilization_ratio"]); @@ -336,6 +336,7 @@ const runtimeObservationReasons = new Set([ "source_not_configured", "sample_timeout", "sample_unavailable", ]); const runtimeProviderTypePattern = /^[a-z][a-z0-9_]{0,31}$/; +const runtimeLifecycleStates = new Set(["active", "sleeping", "transitioning", "pending", "stopped"]); function utf8Length(value: string): number { return new TextEncoder().encode(value).length; } @@ -1106,14 +1107,16 @@ function projectRuntimeObservation(value: unknown, expectedSessionId?: string): if ( (isNone && ( value.instance.kind !== "none" || environmentId !== null || value.provider_type !== null || - allocationId !== null || deviceId !== null || connectionGeneration !== null || allocationCreatedAt !== null + allocationId !== null || deviceId !== null || connectionGeneration !== null || allocationCreatedAt !== null || + value.lifecycle_state !== null )) || (isSelfHosted && ( value.instance.kind !== "self_hosted_connection" || environmentId === null || - allocationId !== null || allocationCreatedAt !== null + allocationId !== null || allocationCreatedAt !== null || value.lifecycle_state !== null )) || (isManaged && ( value.instance.kind !== "managed_allocation" || environmentId === null || connectionGeneration !== null || + !runtimeLifecycleStates.has(String(value.lifecycle_state)) || (allocationId === null && (deviceId !== null || allocationCreatedAt !== null)) )) ) return invalidRuntimeObservation(); @@ -1176,6 +1179,7 @@ function projectRuntimeObservation(value: unknown, expectedSessionId?: string): allocation_id: allocationId, device_id: deviceId, connection_generation: connectionGeneration, }, status: value.status, reason: value.reason as RuntimeObservation["reason"], + lifecycle_state: value.lifecycle_state as RuntimeObservation["lifecycle_state"], allocation_created_at: allocationCreatedAt, resolved_at: value.resolved_at, observed_at: observedAt, started_at: startedAt, cpu, memory, } as RuntimeObservation; diff --git a/packages/agents-client/src/types.ts b/packages/agents-client/src/types.ts index 152a98fe6..e038490b4 100644 --- a/packages/agents-client/src/types.ts +++ b/packages/agents-client/src/types.ts @@ -729,6 +729,7 @@ export interface CreateSessionStreamOptions extends StreamOptions { } export type RuntimeObservationStatus = "observed" | "unsupported" | "unavailable"; +export type RuntimeLifecycleState = "active" | "sleeping" | "transitioning" | "pending" | "stopped"; export type RuntimeObservationReason = | "runtime_mode_not_observable" | "allocation_pending" @@ -768,6 +769,7 @@ export interface RuntimeObservedObservation extends RuntimeObservationBase { device_id: string | null; connection_generation: null; }; + lifecycle_state: RuntimeLifecycleState; status: "observed"; reason: null; allocation_created_at: number | null; @@ -787,6 +789,7 @@ export interface RuntimeUnavailableObservation extends RuntimeObservationBase { device_id: string | null; connection_generation: null; }; + lifecycle_state: RuntimeLifecycleState; status: "unavailable"; reason: RuntimeUnavailableReason; allocation_created_at: number | null; @@ -801,6 +804,7 @@ export interface RuntimeNoneObservation extends RuntimeObservationBase { mode: "none"; provider_type: null; instance: { kind: "none"; allocation_id: null; device_id: null; connection_generation: null }; + lifecycle_state: null; status: "unsupported"; reason: "runtime_mode_not_observable"; allocation_created_at: null; @@ -820,6 +824,7 @@ export interface RuntimeSelfHostedObservation extends RuntimeObservationBase { device_id: string | null; connection_generation: string | null; }; + lifecycle_state: null; status: "unsupported"; reason: "runtime_mode_not_observable"; allocation_created_at: null; diff --git a/services/agents-api/internal/api/runtime_observations.go b/services/agents-api/internal/api/runtime_observations.go index 54d061a2d..c79694c5a 100644 --- a/services/agents-api/internal/api/runtime_observations.go +++ b/services/agents-api/internal/api/runtime_observations.go @@ -179,6 +179,11 @@ func runtimeObservationResponse(observation runtimeobs.Observation) (v1.RuntimeO switch observation.Target.Mode { case runtimeobs.ModeManaged: result.Instance.Kind = "managed_allocation" + lifecycleState, err := runtimeLifecycleState(observation.Target.Instance) + if err != nil { + return v1.RuntimeObservation{}, err + } + result.LifecycleState = &lifecycleState if observation.Target.Instance.AllocationID != "" { result.Instance.AllocationID = &observation.Target.Instance.AllocationID } @@ -219,3 +224,30 @@ func runtimeObservationResponse(observation runtimeobs.Observation) (v1.RuntimeO } return result, nil } + +func runtimeLifecycleState(instance runtimeobs.Instance) (string, error) { + switch instance.AllocationState { + case "": + if instance.AllocationID == "" { + return "pending", nil + } + return "", errors.New("invalid Runtime allocation state") + case "creating": + return "pending", nil + case "cleanup_pending", "released": + return "stopped", nil + case "running": + switch instance.ComputePhase { + case "suspended": + return "sleeping", nil + case "quiescing", "suspending", "restoring", "waking": + return "transitioning", nil + case "disabled", "running": + return "active", nil + default: + return "", errors.New("invalid Runtime compute phase") + } + default: + return "", errors.New("invalid Runtime allocation state") + } +} diff --git a/services/agents-api/internal/api/runtime_observations_test.go b/services/agents-api/internal/api/runtime_observations_test.go index 90dd4ab7f..cbbc952d3 100644 --- a/services/agents-api/internal/api/runtime_observations_test.go +++ b/services/agents-api/internal/api/runtime_observations_test.go @@ -94,15 +94,38 @@ func TestRuntimeObservationResponsePreservesObservedZero(t *testing.T) { now := time.Date(2026, 9, 22, 8, 0, 0, 0, time.UTC) sessionID, environmentID := uuid.NewString(), uuid.NewString() value, err := runtimeObservationResponse(runtimeobs.Observation{ - Target: runtimeobs.Target{SessionID: sessionID, EnvironmentID: environmentID, Mode: runtimeobs.ModeManaged, Instance: runtimeobs.Instance{AllocationID: uuid.NewString(), DeviceID: uuid.NewString(), AllocationCreatedAt: now.Add(-time.Hour)}}, + Target: runtimeobs.Target{SessionID: sessionID, EnvironmentID: environmentID, Mode: runtimeobs.ModeManaged, Instance: runtimeobs.Instance{AllocationID: uuid.NewString(), DeviceID: uuid.NewString(), AllocationState: "running", ComputePhase: "running", AllocationCreatedAt: now.Add(-time.Hour)}}, Status: runtimeobs.StatusObserved, ProviderType: "docker", ResolvedAt: now, Sample: &runtimeobs.Sample{ObservedAt: now, CPUUsageSecondsTotal: &zeroCPU, MemoryUsageBytes: &zeroMemory}, }) - if err != nil || value.CPU == nil || value.CPU.UsageSecondsTotal == nil || *value.CPU.UsageSecondsTotal != 0 || value.Memory == nil || value.Memory.UsageBytes == nil || *value.Memory.UsageBytes != 0 { + if err != nil || value.LifecycleState == nil || *value.LifecycleState != "active" || value.CPU == nil || value.CPU.UsageSecondsTotal == nil || *value.CPU.UsageSecondsTotal != 0 || value.Memory == nil || value.Memory.UsageBytes == nil || *value.Memory.UsageBytes != 0 { t.Fatalf("observed zero was lost: %+v %v", value, err) } } +func TestRuntimeLifecycleStateProjectsProviderNeutralPhases(t *testing.T) { + for _, item := range []struct { + state, phase, want string + }{ + {state: "", phase: "", want: "pending"}, + {state: "creating", phase: "disabled", want: "pending"}, + {state: "running", phase: "disabled", want: "active"}, + {state: "running", phase: "running", want: "active"}, + {state: "running", phase: "quiescing", want: "transitioning"}, + {state: "running", phase: "suspending", want: "transitioning"}, + {state: "running", phase: "suspended", want: "sleeping"}, + {state: "running", phase: "restoring", want: "transitioning"}, + {state: "running", phase: "waking", want: "transitioning"}, + {state: "cleanup_pending", phase: "disabled", want: "stopped"}, + {state: "released", phase: "disabled", want: "stopped"}, + } { + got, err := runtimeLifecycleState(runtimeobs.Instance{AllocationState: item.state, ComputePhase: item.phase}) + if err != nil || got != item.want { + t.Fatalf("state=%s phase=%s got=%s want=%s err=%v", item.state, item.phase, got, item.want, err) + } + } +} + func TestRuntimeObservationResponseRejectsTimesOutsidePublicContract(t *testing.T) { now := time.Date(2026, 9, 22, 8, 0, 0, 0, time.UTC) preEpoch := time.Unix(-1, 0).UTC() From 50bd903dba54850ae7ad5bed228708d38af90fff Mon Sep 17 00:00:00 2001 From: sam Date: Wed, 23 Sep 2026 22:58:36 +0800 Subject: [PATCH 3/4] Keep Runtime trend identities metric-scoped --- .../dashboard/RuntimeTrendCharts.test.tsx | 25 ++++++++++++++++++- .../features/dashboard/RuntimeTrendCharts.tsx | 2 +- 2 files changed, 25 insertions(+), 2 deletions(-) diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx index a2bfff8e7..ebb6a7249 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx @@ -8,10 +8,15 @@ function sample(sampledAt: number, cpuRatio: number | null): RuntimeTrendSample return { sampledAt, targets: [{ + seriesId: "activity:session-1", + label: "Runtime worker", + cpuRatio: null, + runtimeActive: 1, + }, { seriesId: "session-1:allocation-1", label: "Runtime worker", cpuRatio, - runtimeActive: 1, + runtimeActive: null, }], cpuCandidates: [], memoryUsageBytes: 512, @@ -119,6 +124,24 @@ describe("Runtime live-window chart accessibility", () => { expect(html).toContain("1 active / 0 inactive"); }); + it("does not mix Session activity and allocation CPU series identities", () => { + const html = renderToStaticMarkup( + , + ); + + expect(html.match(/aria-label="Hide Runtime worker series"/g)).toHaveLength(2); + + const pending = sample(180_000, null); + pending.targets = [{ + seriesId: "activity:pending-session", + label: "Pending worker", + cpuRatio: null, + runtimeActive: 0, + }]; + const pendingHtml = renderToStaticMarkup(); + expect(pendingHtml.match(/aria-label="Hide Pending worker series"/g)).toHaveLength(1); + }); + it("announces an isolated durable value as sparse rather than empty", () => { const html = renderToStaticMarkup( , diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx index 09029e9d1..ec70b12e1 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx @@ -488,7 +488,7 @@ function targetIds(samples: readonly RuntimeTrendSample[], field: "cpuRatio" | " for (const sample of samples) { for (const target of sample.targets) { const value = target[field]; - latest.set(target.seriesId, value ?? latest.get(target.seriesId) ?? 0); + if (value !== null) latest.set(target.seriesId, value); } } return [...latest.entries()].sort((left, right) => right[1] - left[1]).slice(0, 3).map(([id]) => id); From 890b2d1ad7461f4cfc9687de2b4cabc3a1c8eecf Mon Sep 17 00:00:00 2001 From: sam Date: Thu, 24 Sep 2026 00:23:30 +0800 Subject: [PATCH 4/4] Aggregate active Sandbox trends --- .../dashboard/RuntimeTrendCharts.test.tsx | 19 +++++++++- .../features/dashboard/RuntimeTrendCharts.tsx | 35 +++++++++++++++---- .../features/dashboard/RuntimeTrendPanel.tsx | 4 ++- .../src/features/sessions/SessionsView.tsx | 1 + contracts/agents-api/runtime-history-api.md | 9 +++-- .../runtime-observability-design.md | 4 ++- contracts/agents-api/runtime-observability.md | 11 +++--- 7 files changed, 66 insertions(+), 17 deletions(-) diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx index 49d929e77..eeba94ed4 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx @@ -1,7 +1,7 @@ import { renderToStaticMarkup } from "react-dom/server"; import { describe, expect, it } from "vitest"; -import { RuntimeTrendCharts, runtimeChartCaption, runtimeChartShowsSparsePoints } from "./RuntimeTrendCharts"; +import { integerTickRatios, RuntimeTrendCharts, runtimeChartCaption, runtimeChartShowsSparsePoints } from "./RuntimeTrendCharts"; import type { RuntimeTrendSample } from "./runtime-trends"; function sample(sampledAt: number, cpuRatio: number | null): RuntimeTrendSample { @@ -23,6 +23,11 @@ function sample(sampledAt: number, cpuRatio: number | null): RuntimeTrendSample } describe("Runtime live-window chart accessibility", () => { + it("uses exact integer y-axis positions for Sandbox counts", () => { + expect(integerTickRatios(5).map((ratio) => ratio * 5)).toEqual([5, 3, 2, 0]); + expect(integerTickRatios(17).map((ratio) => ratio * 17)).toEqual([17, 11, 6, 0]); + }); + it("shows isolated or sparse values as points without inventing continuity", () => { expect(runtimeChartShowsSparsePoints([null, 512, null])).toBe(true); expect(runtimeChartShowsSparsePoints([512, 768])).toBe(true); @@ -131,6 +136,18 @@ describe("Runtime live-window chart accessibility", () => { expect(html).toContain("active30"); expect(html.match(/aria-label="Hide active series"/g)).toHaveLength(1); }); + + it("renders the Session view as one binary Runtime-active series", () => { + const latest = { ...sample(120_000, .5), activeSandboxCount: 3 }; + const html = renderToStaticMarkup( + , + ); + + expect(html).toContain("Runtime active"); + expect(html).toContain("1 active / 0 inactive"); + expect(html).toContain("RuntimeActive0"); + expect(html).not.toContain("Active sandboxes"); + }); it("announces an isolated durable value as sparse rather than empty", () => { const html = renderToStaticMarkup( , diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx index 99fbc9797..2a2985ecd 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx @@ -504,16 +504,26 @@ function targetLabel(samples: readonly RuntimeTrendSample[], id: string): string const tones: TrendSeries["tone"][] = ["orange", "green", "blue"]; +export function integerTickRatios(maximum: number): number[] { + const integerMaximum = Math.max(1, Math.ceil(maximum)); + const values = integerMaximum <= 4 + ? Array.from({ length: integerMaximum + 1 }, (_, index) => integerMaximum - index) + : [integerMaximum, Math.round(integerMaximum * 2 / 3), Math.round(integerMaximum / 3), 0]; + return [...new Set(values)].map((value) => value / integerMaximum); +} + export function RuntimeTrendCharts({ samples, source = "live", rangeStart, rangeEnd, + activeDisplay = "sum", }: { samples: readonly RuntimeTrendSample[]; source?: RuntimeTrendSource; rangeStart?: number; rangeEnd?: number; + activeDisplay?: "sum" | "binary"; }) { const charts = useMemo(() => { const cpuIds = targetIds(samples, "cpuRatio"); @@ -530,10 +540,15 @@ export function RuntimeTrendCharts({ const memoryLimit = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryLimitBytes ?? 0 })); const active = [{ id: "active", - label: "active", + label: activeDisplay === "binary" ? "Runtime" : "active", tone: "green", stepped: true, - points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.activeSandboxCount ?? 0 })), + points: samples.map((sample) => ({ + sampledAt: sample.sampledAt, + value: activeDisplay === "binary" + ? (sample.activeSandboxCount ?? 0) > 0 ? 1 : 0 + : sample.activeSandboxCount ?? 0, + })), }] satisfies TrendSeries[]; const throughput = tokenThroughput(samples); return { @@ -548,23 +563,29 @@ export function RuntimeTrendCharts({ { id: "output", label: "output", tone: "green", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.outputPerMinute ?? 0 })) }, ] satisfies TrendSeries[], }; - }, [samples]); + }, [activeDisplay, samples]); const cpuMaximum = Math.max(100, ...finite(charts.cpu.flatMap((series) => series.points.map((point) => point.value)))); const memoryMaximum = Math.max(1, ...finite(charts.memory.flatMap((series) => series.points.map((point) => point.value)))); const activeMaximum = Math.max(1, ...finite(charts.active.flatMap((series) => series.points.map((point) => point.value)))); - const activeTicks = activeMaximum <= 4 - ? Array.from({ length: activeMaximum + 1 }, (_, index) => (activeMaximum - index) / activeMaximum) - : [1, .66, .33, 0]; + const activeTicks = integerTickRatios(activeMaximum); const tokenMaximum = Math.max(1, ...finite(charts.tokens.flatMap((series) => series.points.map((point) => point.value)))); const newest = rangeEnd ?? samples.at(-1)?.sampledAt ?? Date.now(); const oldest = rangeStart ?? samples[0]?.sampledAt ?? newest - 60 * 60 * 1_000; const durable = source === "durable"; + const binaryActive = activeDisplay === "binary"; + const activeTitle = binaryActive ? "Runtime active" : "Active sandboxes"; + const activeSubtitle = binaryActive + ? durable ? "observed allocation in retained bucket · 1 active / 0 inactive" : "lifecycle state active · 1 active / 0 inactive" + : durable ? "observed allocations per retained bucket · durable history" : "lifecycle state active allocations per snapshot · live window"; + const formatActive = binaryActive + ? (value: number) => value >= .5 ? "Active" : "Inactive" + : (value: number) => `${Math.round(value)}`; return (
`${Math.round(value)}%`} rangeStart={oldest} rangeEnd={newest} source={source} bands={[{ from: 0, to: 30, tone: "safe" }, { from: 30, to: 70, tone: "warning" }, { from: 70, to: 100, tone: "danger" }]} ticks={[1, .7, .3, 0]} emptyMessage={durable ? "No retained CPU samples" : undefined} /> formatDashboardBytes(Math.round(value))} rangeStart={oldest} rangeEnd={newest} source={source} emptyMessage={durable ? "No retained observed memory samples" : undefined} /> - `${Math.round(value)}`} rangeStart={oldest} rangeEnd={newest} source={source} ticks={activeTicks} emptyMessage={durable ? "No retained active Sandbox samples" : undefined} /> + `${formatDashboardTokens(Math.round(value))}/min`} rangeStart={oldest} rangeEnd={newest} source={source} emptyMessage={durable ? "No retained token samples" : undefined} />
); diff --git a/apps/web/src/features/dashboard/RuntimeTrendPanel.tsx b/apps/web/src/features/dashboard/RuntimeTrendPanel.tsx index dd7190d40..4ea12a227 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendPanel.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendPanel.tsx @@ -30,6 +30,7 @@ export function RuntimeTrendPanel({ headingId = "dashboard-runtime-live-heading", title = "Resource trends", allowSourceSelection = false, + activeDisplay = "sum", }: { snapshot: RuntimeDashboardSnapshot; stale: boolean; @@ -37,6 +38,7 @@ export function RuntimeTrendPanel({ headingId?: string; title?: string; allowSourceSelection?: boolean; + activeDisplay?: "sum" | "binary"; }) { const [trendSamples, setTrendSamples] = useState(() => appendRuntimeTrendSample([], snapshot)); const [selectedTrendRange, setSelectedTrendRange] = useState(RUNTIME_TREND_WINDOW_MS); @@ -172,7 +174,7 @@ export function RuntimeTrendPanel({ {durableState === "failed" && durableError ?

Durable history refresh failed: {durableError}

: null} {durableState === "unavailable" ?

Durable history is not configured; Live samples remain available.

: null} - + ); } diff --git a/apps/web/src/features/sessions/SessionsView.tsx b/apps/web/src/features/sessions/SessionsView.tsx index 4795abdfb..d3d8c5822 100644 --- a/apps/web/src/features/sessions/SessionsView.tsx +++ b/apps/web/src/features/sessions/SessionsView.tsx @@ -954,6 +954,7 @@ export function SessionsView({ headingId={`session-runtime-trends-heading-${sessionId}`} title="Session resource trends" allowSourceSelection + activeDisplay="binary" />
) : null; diff --git a/contracts/agents-api/runtime-history-api.md b/contracts/agents-api/runtime-history-api.md index 5acda837e..f22337762 100644 --- a/contracts/agents-api/runtime-history-api.md +++ b/contracts/agents-api/runtime-history-api.md @@ -172,6 +172,9 @@ or malformed data reject the entire response with a 502 client projection error. Compute uptime is available from current observations only. Retained allocation series can span compute restarts and unavailable intervals; their earliest start is not a per-bucket compute start and must not be used to draw an uptime history. -Clients may project a binary Runtime activity series from bucket coverage: -`observed_count > 0` is `1`; a missing or unavailable bucket is `0`. This does -not distinguish a sleeping Runtime from a collection failure. +Clients may project a Dashboard active-Sandbox count by counting distinct +allocation identities with `observed_count > 0` in each bucket and deduplicating +the same allocation across Session histories. The single-Session presentation +collapses any positive count to `1`; a missing or unavailable bucket is currently +rendered as `0`. This temporary zero-fill policy does not distinguish a sleeping +Runtime from missing collection coverage. diff --git a/contracts/agents-api/runtime-observability-design.md b/contracts/agents-api/runtime-observability-design.md index 97f47fb1d..c892c017d 100644 --- a/contracts/agents-api/runtime-observability-design.md +++ b/contracts/agents-api/runtime-observability-design.md @@ -388,7 +388,9 @@ acceptance. - Confirmed active Sandbox count over time. Each bucket counts managed allocations with an observed provider sample; unavailable or timed-out samples are not presented as confirmed active. A bucket with collection coverage but no observed - allocation is zero; a bucket without collection coverage remains a gap. + allocation is zero. The history model retains missing coverage as null; the + current Dashboard presentation renders that null as zero until sleeping and + collection-failure history are represented separately. - Data freshness and source coverage. Aggregates include only present measurements. Each total states its denominator, diff --git a/contracts/agents-api/runtime-observability.md b/contracts/agents-api/runtime-observability.md index bc7659c0b..36cc5938d 100644 --- a/contracts/agents-api/runtime-observability.md +++ b/contracts/agents-api/runtime-observability.md @@ -79,10 +79,13 @@ This phase supplies compute uptime evidence and retains the existing durable allocation and Turn timestamps. It does not infer idle time. CPU quietness, heartbeat age, connection status, and `kept_at` are not authoritative idle state. -Web projects Runtime activity as a binary chart: a successful observation in a -time bucket is `1`; an absent or unavailable observation is `0`. This operational -availability view is intentionally not a durable classification of sleeping -versus collection failure. +Web projects active Runtime state differently by scope. The Dashboard shows one +summed series of distinct allocation identities: live snapshots count +`lifecycle_state: active`, while retained buckets count successfully observed +allocations because lifecycle state is not retained yet. The single-Session view +collapses the same value to `1` or `0`. Missing or unavailable retained values are +currently rendered as zero, so this presentation intentionally does not yet +distinguish sleeping from collection failure. Future automatic suspension requires a separate durable control model, including an activity revision and timestamps such as `idle_since` and