From a235adcdc5f0458c9601a8ed25557a42fa64b785 Mon Sep 17 00:00:00 2001 From: sam Date: Wed, 23 Sep 2026 20:19:17 +0800 Subject: [PATCH] Show active sandbox count and observed memory --- apps/web/e2e/agents-lifecycle.spec.ts | 9 +-- .../features/dashboard/DashboardView.test.tsx | 1 + .../dashboard/RuntimeTrendCharts.test.tsx | 11 ++-- .../features/dashboard/RuntimeTrendCharts.tsx | 28 +++++----- .../dashboard/runtime-history.test.ts | 55 +++++++++++++++++-- .../src/features/dashboard/runtime-history.ts | 31 ++++++++--- .../features/dashboard/runtime-trends.test.ts | 24 ++++++++ .../src/features/dashboard/runtime-trends.ts | 16 +++++- .../runtime-observability-design.md | 17 ++++-- 9 files changed, 151 insertions(+), 41 deletions(-) diff --git a/apps/web/e2e/agents-lifecycle.spec.ts b/apps/web/e2e/agents-lifecycle.spec.ts index 2dd5df09a..7b4b5626b 100644 --- a/apps/web/e2e/agents-lifecycle.spec.ts +++ b/apps/web/e2e/agents-lifecycle.spec.ts @@ -2680,7 +2680,7 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand await expect(dashboard.locator(".dashboard-runtime-sample-count")).toContainText("1 sample ·"); await expect(dashboard.getByRole("heading", { name: "CPU usage" })).toBeVisible(); await expect(dashboard.getByRole("heading", { name: "Memory usage" })).toBeVisible(); - await expect(dashboard.getByRole("heading", { name: "Compute uptime" })).toBeVisible(); + await expect(dashboard.getByRole("heading", { name: "Active Sandboxes" })).toBeVisible(); await expect(dashboard.getByRole("heading", { name: "Token throughput" })).toBeVisible(); await expect(dashboard.getByLabel("Live Runtime sampling every 30 seconds")).toBeVisible(); const liveRange = dashboard.getByRole("group", { name: "Runtime live range" }); @@ -2692,7 +2692,7 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand await refresh.click(); await expect(dashboard.getByLabel("CPU usage: 3 live samples")).toBeVisible(); await expect(dashboard.getByLabel("Memory usage: 3 live samples")).toBeVisible(); - await expect(dashboard.getByLabel("Compute uptime: 3 live samples")).toBeVisible(); + await expect(dashboard.getByLabel("Active Sandboxes: 3 live samples")).toBeVisible(); await expect(dashboard.getByLabel("Token throughput: 3 live samples")).toBeVisible(); await expect(dashboard.locator(".dashboard-runtime-sample-count")).toContainText("3 samples"); await expect(dashboard.getByText("CPU usage live trend available")).toBeAttached(); @@ -2760,7 +2760,7 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand const keyboardSelectedAt = Number(await cpuChart.getAttribute("data-selected-at")); expect(keyboardSelectedAt).toBeGreaterThanOrEqual(zoomedViewStart); expect(keyboardSelectedAt).toBeLessThanOrEqual(zoomedViewEnd); - for (const chartName of ["Memory usage", "Compute uptime", "Token throughput"]) { + for (const chartName of ["Memory usage", "Active Sandboxes", "Token throughput"]) { await expect(dashboard.getByLabel(`${chartName}: 3 live samples`)).toHaveAttribute("data-view-start", String(initialViewStart)); await expect(dashboard.getByLabel(`${chartName}: 3 live samples`)).toHaveAttribute("data-view-end", String(initialViewEnd)); } @@ -2910,10 +2910,11 @@ test("restores retained Runtime history after a Dashboard reload", async ({ page await expect(dashboard.getByLabel(/Durable · 30s; 1 Runtime targets/)).toBeVisible(); await expect(dashboard.getByLabel("Runtime durable-history charts")).toBeVisible(); await expect(dashboard.getByRole("heading", { name: "Compute uptime", exact: true })).toHaveCount(0); - await expect(dashboard.locator('[data-chart-engine="uplot"]')).toHaveCount(3); + await expect(dashboard.locator('[data-chart-engine="uplot"]')).toHaveCount(4); await expect(dashboard.getByText("CPU usage durable trend available")).toBeAttached(); await expect(dashboard).toContainText("120 buckets"); await expect(dashboard).toContainText("119/120 observations"); + await expect(dashboard.getByText("Active Sandboxes durable trend available")).toBeAttached(); await expect(dashboard.getByText("Token throughput durable trend available")).toBeAttached(); const durableCpuChart = dashboard.getByLabel("CPU usage: 120 retained buckets"); await expect(dashboard.getByRole("region", { name: "CPU usage durable history chart" })).toBeVisible(); diff --git a/apps/web/src/features/dashboard/DashboardView.test.tsx b/apps/web/src/features/dashboard/DashboardView.test.tsx index 3cb3e67a7..a3ec29127 100644 --- a/apps/web/src/features/dashboard/DashboardView.test.tsx +++ b/apps/web/src/features/dashboard/DashboardView.test.tsx @@ -257,6 +257,7 @@ describe("Dashboard loaded-result presentation", () => { expect(html).toContain("CPU usage"); expect(html).toContain("Memory usage"); expect(html).not.toContain("Compute uptime"); + expect(html).toContain("Active Sandboxes"); expect(html).toContain("Token throughput"); expect(html).toContain("No retained CPU samples"); expect(html).toContain("0/2 valid points · 0 snapshots · no history is synthesized"); diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx index c29e9be82..80052f604 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx @@ -7,6 +7,7 @@ import type { RuntimeTrendSample } from "./runtime-trends"; function sample(sampledAt: number, cpuRatio: number | null): RuntimeTrendSample { return { sampledAt, + activeSandboxCount: 1, targets: [{ seriesId: "session-1:allocation-1", label: "Runtime worker", @@ -38,7 +39,7 @@ describe("Runtime live-window chart accessibility", () => { allSeriesHidden: true, validPoints: 0, sampleCount: 24, - emptyMessage: "No complete retained memory samples", + emptyMessage: "No retained observed memory samples", })).toBe("Memory usage all series hidden; use the legend to show a series"); }); @@ -58,7 +59,7 @@ describe("Runtime live-window chart accessibility", () => { expect(html.match(/data-chart-engine="uplot"/g)).toHaveLength(4); expect(html).not.toContain("Collecting live samples"); - expect(html).toContain("Compute uptime"); + expect(html).toContain("Active Sandboxes"); }); it("exposes interactive series, point selection, and Grafana-style in-plot range selection", () => { @@ -82,9 +83,11 @@ describe("Runtime live-window chart accessibility", () => { ); expect(html).toContain('aria-label="CPU usage durable history chart"'); - expect(html.match(/data-chart-engine="uplot"/g)).toHaveLength(3); + expect(html.match(/data-chart-engine="uplot"/g)).toHaveLength(4); expect(html).not.toContain("Compute uptime"); expect(html).toContain('aria-label="CPU usage: 2 retained buckets"'); + expect(html).toContain('aria-label="Active Sandboxes durable history chart"'); + expect(html).toContain("active10"); expect(html).not.toContain('aria-label="CPU usage: 2 live samples"'); }); @@ -94,6 +97,6 @@ describe("Runtime live-window chart accessibility", () => { ); expect(html).toContain("Memory usage durable trend has 1 sparse valid point; a line requires consecutive buckets"); - expect(html).not.toContain("Memory usage No complete retained memory samples"); + expect(html).not.toContain("Memory usage No retained observed memory samples"); }); }); diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx index 2edf56ca3..521e4e18a 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx @@ -9,7 +9,7 @@ import { import uPlot from "uplot"; import "uplot/dist/uPlot.min.css"; -import { formatDashboardBytes, formatDashboardDuration, formatDashboardTokens } from "./dashboard-model"; +import { formatDashboardBytes, formatDashboardTokens } from "./dashboard-model"; import { tokenThroughput, type RuntimeTrendSample } from "./runtime-trends"; interface TrendPoint { @@ -481,7 +481,7 @@ function TrendChart({ ); } -function targetIds(samples: readonly RuntimeTrendSample[], field: "cpuRatio" | "uptimeSeconds"): string[] { +function targetIds(samples: readonly RuntimeTrendSample[], field: "cpuRatio"): string[] { const latest = new Map(); for (const sample of samples) { for (const target of sample.targets) { @@ -515,7 +515,6 @@ export function RuntimeTrendCharts({ }) { const charts = useMemo(() => { const cpuIds = targetIds(samples, "cpuRatio"); - const uptimeIds = targetIds(samples, "uptimeSeconds"); const cpu = cpuIds.map((id, index): TrendSeries => ({ id, label: targetLabel(samples, id), @@ -524,12 +523,12 @@ export function RuntimeTrendCharts({ })); const memoryUsed = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryUsageBytes })); const memoryLimit = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryLimitBytes })); - const uptime = uptimeIds.map((id, index): TrendSeries => ({ - id, - label: targetLabel(samples, id), - tone: tones[(index + 2) % tones.length] ?? "blue", - points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.targets.find((target) => target.seriesId === id)?.uptimeSeconds ?? null })), - })); + const active = [{ + id: "active", + label: "active", + tone: "green", + points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.activeSandboxCount })), + }] satisfies TrendSeries[]; const throughput = tokenThroughput(samples); return { cpu, @@ -537,7 +536,7 @@ export function RuntimeTrendCharts({ { id: "used", label: "used", tone: "purple", points: memoryUsed }, { id: "limit", label: "configured limit", tone: "green", points: memoryLimit }, ] satisfies TrendSeries[], - uptime, + active, tokens: [ { id: "input", label: "input", tone: "orange", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.inputPerMinute })) }, { id: "output", label: "output", tone: "green", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.outputPerMinute })) }, @@ -546,7 +545,10 @@ export function RuntimeTrendCharts({ }, [samples]); const cpuMaximum = Math.max(100, ...finite(charts.cpu.flatMap((series) => series.points.map((point) => point.value)))); const memoryMaximum = Math.max(1, ...finite(charts.memory.flatMap((series) => series.points.map((point) => point.value)))); - const uptimeMaximum = Math.max(1, ...finite(charts.uptime.flatMap((series) => series.points.map((point) => point.value)))); + const activeMaximum = Math.max(1, ...finite(charts.active.flatMap((series) => series.points.map((point) => point.value)))); + const activeTicks = activeMaximum <= 4 + ? Array.from({ length: activeMaximum + 1 }, (_, index) => (activeMaximum - index) / activeMaximum) + : [1, .66, .33, 0]; const tokenMaximum = Math.max(1, ...finite(charts.tokens.flatMap((series) => series.points.map((point) => point.value)))); const newest = rangeEnd ?? samples.at(-1)?.sampledAt ?? Date.now(); const oldest = rangeStart ?? samples[0]?.sampledAt ?? newest - 60 * 60 * 1_000; @@ -555,8 +557,8 @@ export function RuntimeTrendCharts({ return (
`${Math.round(value)}%`} rangeStart={oldest} rangeEnd={newest} source={source} bands={[{ from: 0, to: 30, tone: "safe" }, { from: 30, to: 70, tone: "warning" }, { from: 70, to: 100, tone: "danger" }]} ticks={[1, .7, .3, 0]} emptyMessage={durable ? "No retained CPU samples" : undefined} /> - formatDashboardBytes(Math.round(value))} rangeStart={oldest} rangeEnd={newest} source={source} emptyMessage={durable ? "No complete retained memory samples" : undefined} /> - {!durable ? formatDashboardDuration(value)} rangeStart={oldest} rangeEnd={newest} source={source} /> : null} + formatDashboardBytes(Math.round(value))} rangeStart={oldest} rangeEnd={newest} source={source} emptyMessage={durable ? "No retained observed memory samples" : undefined} /> + `${Math.round(value)}`} rangeStart={oldest} rangeEnd={newest} source={source} ticks={activeTicks} emptyMessage={durable ? "No confirmed active Sandbox samples" : undefined} /> `${formatDashboardTokens(Math.round(value))}/min`} rangeStart={oldest} rangeEnd={newest} source={source} emptyMessage={durable ? "No retained token samples" : undefined} />
); diff --git a/apps/web/src/features/dashboard/runtime-history.test.ts b/apps/web/src/features/dashboard/runtime-history.test.ts index 58cbfcd3c..d60ede3d1 100644 --- a/apps/web/src/features/dashboard/runtime-history.test.ts +++ b/apps/web/src/features/dashboard/runtime-history.test.ts @@ -125,6 +125,7 @@ describe("Runtime Durable Dashboard history", () => { expect(samples).toHaveLength(2); expect(samples[0]).toMatchObject({ sampledAt: 130_000, + activeSandboxCount: 1, memoryUsageBytes: 512, memoryLimitBytes: 1_024, inputTokensPerMinute: null, @@ -135,6 +136,49 @@ describe("Runtime Durable Dashboard history", () => { expect(samples[1]).toMatchObject({ inputTokensPerMinute: 60, outputTokensPerMinute: 20 }); }); + it("counts and aggregates distinct Runtime allocations within one Session", () => { + const source = history(); + const secondSeries = { + ...source.series[0]!, + allocation_id: "55555555-5555-4555-8555-555555555555", + points: source.series[0]!.points.map((point) => ({ + ...point, + memory: point.memory ? { ...point.memory, usage_bytes: 128, limit_bytes: 256 } : null, + })), + }; + const samples = runtimeDurableTrendSamples([session], [{ + ...source, + series: [source.series[0]!, secondSeries], + }]); + + expect(samples[0]).toMatchObject({ + activeSandboxCount: 2, + memoryUsageBytes: 640, + memoryLimitBytes: 1_280, + }); + }); + + it("deduplicates one Runtime allocation repeated across Session histories", () => { + const second = { ...session, id: "44444444-4444-4444-8444-444444444444" } as AgentSession; + const repeated = history({ + session_id: second.id, + series: [{ + ...history().series[0]!, + points: history().series[0]!.points.map((point) => ({ + ...point, + memory: point.memory ? { ...point.memory, usage_bytes: 128, limit_bytes: 256 } : null, + })), + }], + }); + const samples = runtimeDurableTrendSamples([session, second], [history(), repeated]); + + expect(samples[0]).toMatchObject({ + activeSandboxCount: 1, + memoryUsageBytes: 128, + memoryLimitBytes: 256, + }); + }); + it("does not derive compute uptime from retained allocation starts or unavailable observations", () => { const source = history(); source.series[0]!.points[1] = { @@ -142,9 +186,10 @@ describe("Runtime Durable Dashboard history", () => { }; const samples = runtimeDurableTrendSamples([session], [source]); expect(samples.flatMap((sample) => sample.targets.map((target) => target.uptimeSeconds))).toEqual([null, null]); + expect(samples[1]?.activeSandboxCount).toBe(0); }); - it("keeps aggregate memory absent when any queried target has no memory value", () => { + it("aggregates observed memory without letting an unavailable target erase it", () => { const second = { ...session, id: "44444444-4444-4444-8444-444444444444" } as AgentSession; const secondHistory = history({ session_id: second.id, @@ -152,7 +197,8 @@ describe("Runtime Durable Dashboard history", () => { series: [], }); const samples = runtimeDurableTrendSamples([session, second], [history(), secondHistory]); - expect(samples.every((sample) => sample.memoryUsageBytes === null && sample.memoryLimitBytes === null)).toBe(true); + expect(samples.every((sample) => sample.memoryUsageBytes !== null && sample.memoryLimitBytes !== null)).toBe(true); + expect(samples[0]).toMatchObject({ memoryUsageBytes: 512, memoryLimitBytes: 1_024, activeSandboxCount: 1 }); }); it("keeps omitted buckets between distant observations as gaps", () => { @@ -189,7 +235,7 @@ describe("Runtime Durable Dashboard history", () => { expect(samples[10]?.outputTokensPerMinute).toBeNull(); for (const sample of samples.slice(1, -1)) { expect(sample).toMatchObject({ - targets: [], memoryUsageBytes: null, memoryLimitBytes: null, + activeSandboxCount: null, targets: [], memoryUsageBytes: null, memoryLimitBytes: null, inputTokensPerMinute: null, outputTokensPerMinute: null, }); } @@ -203,6 +249,7 @@ describe("Runtime Durable Dashboard history", () => { expect(samples.map((sample) => sample.sampledAt)).toEqual([100_000, 130_000, 160_000, 190_000, 205_000]); expect(samples.map((sample) => sample.memoryUsageBytes)).toEqual([null, 512, 768, null, null]); expect(samples.map((sample) => sample.targets.length)).toEqual([0, 1, 1, 0, 0]); + expect(samples.map((sample) => sample.activeSandboxCount)).toEqual([null, 1, 1, null, null]); }); it("represents an entirely missing range without fabricating zero measurements", () => { @@ -219,7 +266,7 @@ describe("Runtime Durable Dashboard history", () => { expect(samples.map((sample) => sample.sampledAt)).toEqual([130_000, 160_000, 175_000]); for (const sample of samples) { expect(sample).toMatchObject({ - targets: [], memoryUsageBytes: null, memoryLimitBytes: null, + activeSandboxCount: null, targets: [], memoryUsageBytes: null, memoryLimitBytes: null, inputTokensPerMinute: null, outputTokensPerMinute: null, }); } diff --git a/apps/web/src/features/dashboard/runtime-history.ts b/apps/web/src/features/dashboard/runtime-history.ts index 9200bf92a..093f472c0 100644 --- a/apps/web/src/features/dashboard/runtime-history.ts +++ b/apps/web/src/features/dashboard/runtime-history.ts @@ -80,7 +80,9 @@ async function mapBounded( interface MutableBucket { sampledAt: number; + hasObservationCoverage: boolean; targets: Map; + activeSandboxes: Map; memory: Map; tokens: Map; } @@ -94,7 +96,7 @@ export function runtimeDurableTrendSamples( const bucket = (sampledAt: number): MutableBucket => { let value = buckets.get(sampledAt); if (!value) { - value = { sampledAt, targets: new Map(), memory: new Map(), tokens: new Map() }; + value = { sampledAt, hasObservationCoverage: false, targets: new Map(), activeSandboxes: new Map(), memory: new Map(), tokens: new Map() }; buckets.set(sampledAt, value); } return value; @@ -105,7 +107,10 @@ export function runtimeDurableTrendSamples( for (let bucketStart = start; bucketStart < end; bucketStart += history.resolution_seconds) { bucket(Math.min(bucketStart + history.resolution_seconds, end) * 1_000); } - for (const coverage of history.coverage.buckets) bucket(coverage.end * 1_000); + for (const coverage of history.coverage.buckets) { + const value = bucket(coverage.end * 1_000); + value.hasObservationCoverage ||= coverage.observation_count > 0; + } for (const usage of history.token_usage) { bucket(usage.end * 1_000).tokens.set(history.session_id, { sampledAt: usage.sampled_at * 1_000, @@ -118,6 +123,7 @@ export function runtimeDurableTrendSamples( const label = titles.get(history.session_id) ?? "Runtime"; for (const point of series.points) { const value = bucket(point.end * 1_000); + value.hasObservationCoverage ||= point.observation_count > 0; const observedAt = point.last_observed_at; value.targets.set(targetID, { seriesId: targetID, @@ -125,12 +131,18 @@ export function runtimeDurableTrendSamples( cpuRatio: point.cpu?.utilization_ratio ?? null, uptimeSeconds: null, }); + if (observedAt !== null && point.observed_count > 0) { + const previous = value.activeSandboxes.get(series.allocation_id); + if (!previous || observedAt >= previous.observedAt) { + value.activeSandboxes.set(series.allocation_id, { observedAt }); + } + } const usage = point.memory?.usage_bytes; const limit = point.memory?.limit_bytes; if (observedAt !== null && usage != null && limit != null) { - const previous = value.memory.get(history.session_id); + const previous = value.memory.get(series.allocation_id); if (!previous || observedAt >= previous.observedAt) { - value.memory.set(history.session_id, { observedAt, usage, limit }); + value.memory.set(series.allocation_id, { observedAt, usage, limit }); } } } @@ -138,16 +150,17 @@ export function runtimeDurableTrendSamples( } const samples = [...buckets.values()].sort((left, right) => left.sampledAt - right.sampledAt).map((value) => { - const completeMemory = sessions.length > 0 && value.memory.size === sessions.length; + const observedMemory = [...value.memory.values()]; return { sampledAt: value.sampledAt, + activeSandboxCount: value.hasObservationCoverage ? value.activeSandboxes.size : null, targets: [...value.targets.values()], cpuCandidates: [], - memoryUsageBytes: completeMemory - ? [...value.memory.values()].reduce((total, current) => total + current.usage, 0) + memoryUsageBytes: observedMemory.length > 0 + ? observedMemory.reduce((total, current) => total + current.usage, 0) : null, - memoryLimitBytes: completeMemory - ? [...value.memory.values()].reduce((total, current) => total + current.limit, 0) + memoryLimitBytes: observedMemory.length > 0 + ? observedMemory.reduce((total, current) => total + current.limit, 0) : null, tokenTotals: value.tokens.size === sessions.length ? [...value.tokens.entries()].map(([sessionId, usage]) => ({ sessionId, ...usage })) diff --git a/apps/web/src/features/dashboard/runtime-trends.test.ts b/apps/web/src/features/dashboard/runtime-trends.test.ts index 4554c7364..1587e6186 100644 --- a/apps/web/src/features/dashboard/runtime-trends.test.ts +++ b/apps/web/src/features/dashboard/runtime-trends.test.ts @@ -92,6 +92,7 @@ describe("Runtime live-window trends", () => { const sample = runtimeTrendSample(snapshot(120_000)); expect(sample).toMatchObject({ sampledAt: 120_000, + activeSandboxCount: 1, tokenTotals: [{ sessionId: "11111111-1111-4111-8111-111111111111", inputTokens: 100, @@ -107,6 +108,29 @@ describe("Runtime live-window trends", () => { })]); }); + it("deduplicates live aggregate count and memory by Runtime allocation identity", () => { + const duplicate = snapshot(120_000); + const secondSession = { + ...duplicate.sessions[0]!, + id: "44444444-4444-4444-8444-444444444444", + } as AgentSession; + const secondObservation = { + ...duplicate.observations[0]!, + id: secondSession.id, + session_id: secondSession.id, + observed_at: (duplicate.observations[0]!.observed_at ?? 0) + 1, + memory: { usage_bytes: 128, limit_bytes: 256 }, + } as RuntimeObservation; + duplicate.sessions.push(secondSession); + duplicate.observations.push(secondObservation); + + expect(runtimeTrendSample(duplicate)).toMatchObject({ + activeSandboxCount: 1, + memoryUsageBytes: 128, + memoryLimitBytes: 256, + }); + }); + it("deduplicates refreshes and bounds the rolling window", () => { let samples = appendRuntimeTrendSample([], snapshot(60_000), 120_000, 2); samples = appendRuntimeTrendSample(samples, snapshot(120_000)); diff --git a/apps/web/src/features/dashboard/runtime-trends.ts b/apps/web/src/features/dashboard/runtime-trends.ts index 8f9c9ce85..ba675c341 100644 --- a/apps/web/src/features/dashboard/runtime-trends.ts +++ b/apps/web/src/features/dashboard/runtime-trends.ts @@ -30,6 +30,7 @@ export interface RuntimeTrendCPUCandidate extends RuntimeTrendTarget { export interface RuntimeTrendSample { sampledAt: number; + activeSandboxCount: number | null; targets: RuntimeTrendTarget[]; cpuCandidates: RuntimeTrendCPUCandidate[]; memoryUsageBytes: number | null; @@ -110,9 +111,11 @@ export function runtimeTrendSample(snapshot: RuntimeDashboardSnapshot): RuntimeT const session = sessions.get(observation.session_id); if (!session || observation.status !== "observed") return []; const key = allocationKey(observation); - if (key === null) return []; + const allocationId = observation.instance.allocation_id; + if (key === null || typeof allocationId !== "string" || allocationId.length === 0) return []; return [{ seriesId: `${observation.session_id}:${key}`, + allocationId, label: sessionTitle(session), cpuRatio: reportedCpuRatio(observation), observedAt: safeInteger(observation.observed_at), @@ -141,11 +144,20 @@ export function runtimeTrendSample(snapshot: RuntimeDashboardSnapshot): RuntimeT cpuRatio: target.cpuRatio, uptimeSeconds: target.uptimeSeconds, })); - const pairedMemory = observed.filter((target) => ( + const latestByAllocation = new Map(); + for (const target of observed) { + const previous = latestByAllocation.get(target.allocationId); + if (!previous || (target.observedAt ?? -1) >= (previous.observedAt ?? -1)) { + latestByAllocation.set(target.allocationId, target); + } + } + const allocations = [...latestByAllocation.values()]; + const pairedMemory = allocations.filter((target) => ( target.memoryUsageBytes !== null && target.memoryLimitBytes !== null )); return { sampledAt: snapshot.loadedAt, + activeSandboxCount: allocations.length, targets, cpuCandidates: observed.flatMap((target): RuntimeTrendCPUCandidate[] => ( target.cpuRatio !== null || ( diff --git a/contracts/agents-api/runtime-observability-design.md b/contracts/agents-api/runtime-observability-design.md index d55958231..a72a5ffe2 100644 --- a/contracts/agents-api/runtime-observability-design.md +++ b/contracts/agents-api/runtime-observability-design.md @@ -187,10 +187,11 @@ measurement or lifecycle state. | Idle duration | future durable `idle_since` | Not available in the current design. | Container restart resets compute uptime but not allocation age. Live CPU deltas -require the same known compute start as well as the same allocation. Retained -charts show CPU, memory and tokens; uptime stays in the current/Live view because -the history contract does not supply each bucket's compute start. Dashboard labels -must not collapse these values into one generic Runtime duration. +require the same known compute start as well as the same allocation. Trend charts +show CPU, memory, confirmed active Sandbox count, and tokens. Compute uptime stays +in current target details because the history contract does not supply each +bucket's compute start. Dashboard labels must not collapse these values into one +generic Runtime duration. ## 9. Collection behavior @@ -350,7 +351,9 @@ Every bucket reports explicit observation coverage and nullable CPU/memory values. CPU utilization may be derived only from ordered cumulative counters inside one fence; successive intervals are assigned to the bucket containing their right endpoint and combined by CPU-capacity time. Memory uses the final -observed value in the bucket. Empty +observed value in the bucket. Dashboard memory totals aggregate only allocations with +a complete observed usage/limit pair in that bucket; an unavailable or released +target does not erase measurements from active targets. Empty buckets remain gaps. The service rejects cross-scope rows, duplicate series, overlapping or out-of-range buckets, unsafe provider labels, invalid numeric values, and results exceeding the total point budget. @@ -377,6 +380,10 @@ acceptance. - Observed CPU usage and known configured capacity. - Observed memory usage and known limits. - Reported Session tokens, together with the reporting Session count. +- Confirmed active Sandbox count over time. Each bucket counts managed allocations + with an observed provider sample; unavailable or timed-out samples are not + presented as confirmed active. A bucket with collection coverage but no observed + allocation is zero; a bucket without collection coverage remains a gap. - Data freshness and source coverage. Aggregates include only present measurements. Each total states its denominator,