diff --git a/apps/web/e2e/agents-lifecycle.spec.ts b/apps/web/e2e/agents-lifecycle.spec.ts index 9de471d89..7b63cb271 100644 --- a/apps/web/e2e/agents-lifecycle.spec.ts +++ b/apps/web/e2e/agents-lifecycle.spec.ts @@ -2684,6 +2684,11 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand const dashboard = page.locator(".dashboard-page"); await expect(dashboard.getByRole("heading", { name: "Dashboard", exact: true })).toBeVisible(); const refresh = dashboard.getByRole("button", { name: "Refresh Dashboard snapshot" }); + const sandboxDiagnostics = dashboard.getByRole("region", { name: "Sandbox diagnostics" }); + await expect(sandboxDiagnostics).toBeVisible(); + await expect(sandboxDiagnostics).toContainText("12 managed targets"); + await expect(sandboxDiagnostics).toContainText("12 with measured usage and limit"); + await expect(sandboxDiagnostics).toContainText("Highest memory pressure"); await expect(dashboard.locator(".dashboard-runtime-sample-count")).toContainText("1 sample ·"); await expect(dashboard.getByRole("heading", { name: "CPU usage" })).toBeVisible(); await expect(dashboard.getByRole("heading", { name: "Memory usage" })).toBeVisible(); @@ -2735,7 +2740,7 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand await expect(cpuCard.locator(".dashboard-runtime-trend-tooltip")).toBeVisible(); await cpuChart.click({ position: { x: 260, y: 90 } }); await expect(cpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("Pinned"); - await expect(cpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("10%"); + await expect(cpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("Unavailable"); await cpuChart.focus(); await cpuChart.press("ArrowRight"); await expect(cpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("Pinned"); @@ -2930,9 +2935,9 @@ test("restores retained Runtime history after a Dashboard reload", async ({ page const durableCpuCard = durableCpuChart.locator("xpath=ancestor::section[contains(@class, 'dashboard-runtime-trend-card')]"); await durableCpuChart.focus(); await durableCpuChart.press("ArrowLeft"); - await expect(durableCpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("0%"); + await expect(durableCpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("Unavailable"); await durableCpuChart.press("ArrowLeft"); - await expect(durableCpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("0%"); + await expect(durableCpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("Unavailable"); const durableMemoryCard = dashboard.getByRole("region", { name: "Memory usage durable history chart" }); const durableMemorySpan = await durableMemoryCard.locator("canvas").evaluate((canvas: HTMLCanvasElement) => { const context = canvas.getContext("2d"); diff --git a/apps/web/src/features/dashboard/DashboardView.css b/apps/web/src/features/dashboard/DashboardView.css index 994880b84..5364e670c 100644 --- a/apps/web/src/features/dashboard/DashboardView.css +++ b/apps/web/src/features/dashboard/DashboardView.css @@ -633,6 +633,64 @@ border-bottom: 1px solid var(--line); } +.dashboard-sandbox-insights { + padding: 16px; + border-bottom: 1px solid var(--line); +} + +.dashboard-sandbox-insights > header, +.dashboard-sandbox-diagnostics { + display: flex; + justify-content: space-between; + gap: 16px; +} + +.dashboard-sandbox-insights h3, +.dashboard-sandbox-insights h4, +.dashboard-sandbox-insights p { + margin: 0; +} + +.dashboard-sandbox-insights h3 { font-size: 14px; } +.dashboard-sandbox-insights h4 { font-size: 12px; } +.dashboard-sandbox-insights p, +.dashboard-sandbox-insights small { color: var(--fg-muted); font-size: 11px; } + +.dashboard-sandbox-insight-grid { + display: grid; + grid-template-columns: repeat(4, minmax(0, 1fr)); + gap: 8px; + margin-top: 12px; +} + +.dashboard-sandbox-insight-grid > div { + display: grid; + gap: 3px; + padding: 10px 12px; + border: 1px solid var(--line); + border-radius: 8px; + min-width: 0; +} + +.dashboard-sandbox-insight-grid strong { + font: 550 20px/1.2 var(--font-mono); + font-variant-numeric: tabular-nums; +} + +.dashboard-sandbox-insight-grid span { color: var(--fg-muted); font-size: 10px; } +.dashboard-sandbox-diagnostics { margin-top: 14px; } +.dashboard-sandbox-diagnostics > div { flex: 1; min-width: 0; } +.dashboard-sandbox-diagnostics ul { list-style: none; margin: 7px 0 0; padding: 0; } +.dashboard-sandbox-diagnostics li { display: flex; justify-content: space-between; gap: 8px; padding: 4px 0; font-size: 11px; } +.dashboard-sandbox-diagnostics li + li { border-top: 1px solid var(--line); } +.dashboard-sandbox-diagnostics li button { padding: 0; border: 0; background: none; color: var(--accent); cursor: pointer; text-align: left; } +.dashboard-sandbox-diagnostics li span { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } + +@media (max-width: 800px) { + .dashboard-sandbox-insight-grid { grid-template-columns: repeat(2, minmax(0, 1fr)); } + .dashboard-sandbox-diagnostics { flex-direction: column; } +} + .dashboard-runtime-metric { display: grid; min-width: 0; diff --git a/apps/web/src/features/dashboard/DashboardView.test.tsx b/apps/web/src/features/dashboard/DashboardView.test.tsx index 39f3a23e0..4ad132758 100644 --- a/apps/web/src/features/dashboard/DashboardView.test.tsx +++ b/apps/web/src/features/dashboard/DashboardView.test.tsx @@ -247,6 +247,12 @@ describe("Dashboard loaded-result presentation", () => { expect(html).toContain("Cumulative CPU / capacity"); expect(html).toContain("1m 13s / 2 cores"); expect(html).toContain("512 MiB / 2.00 GiB"); + expect(html).toContain("Sandbox diagnostics"); + expect(html).toContain("Memory ≥80%"); + const withoutMemory = render({ + runtimeSnapshot: { sessions: [hosted], observations: [{ ...observation, memory: { usage_bytes: null, limit_bytes: null } }], loadedAt: 1_700_000_100_000 }, + }); + expect(withoutMemory).toMatch(/Memory ≥80%<\/small>Unavailable<\/strong>/); expect(html).toContain('aria-label="Runtime live-window charts"'); expect(html).toContain("Resource trends"); expect(html).toContain("Browser-local samples · reset on reload"); diff --git a/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx b/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx index 2c39d9d37..f40f5e62e 100644 --- a/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx +++ b/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx @@ -26,6 +26,7 @@ import { import { buildRuntimeDashboardModel, + buildSandboxInsights, formatDashboardBytes, formatDashboardDuration, formatDashboardTimestamp, @@ -312,6 +313,9 @@ export function RuntimeObservabilityContent({ [snapshot, tokenTotals], ); const summary = model.summary; + const sandbox = useMemo(() => buildSandboxInsights(model.rows), [model.rows]); + const unavailableReasons = Object.entries(sandbox.unavailableReasons) + .sort((left, right) => right[1] - left[1]); return ( <>
@@ -321,6 +325,34 @@ export function RuntimeObservabilityContent({ } label={t("runtime.metrics.tokens")} value={summary.totalTokens === null ? t("runtime.filters.unavailable") : formatDashboardTokens(summary.totalTokens, locale)} detail={t("runtime.metrics.tokenDetail", { covered: summary.tokenCoverageCount, total: summary.sessionCount })} />
+
+
+
+

{t("sandbox.title")}

+

{t("sandbox.subtitle")}

+
+ {t(stale ? "sandbox.retained" : "sandbox.current")} +
+
+
{t("sandbox.observed")}{sandbox.observed.toLocaleString(locale)}{t("sandbox.observedDetail", { total: summary.managedRuntimeCount })}
+
{t("sandbox.unavailable")}{sandbox.unavailable.toLocaleString(locale)}{t("sandbox.unavailableDetail")}
+
{t("sandbox.highMemory")}{sandbox.measuredMemory === 0 ? t("runtime.filters.unavailable") : sandbox.highMemory.toLocaleString(locale)}{t("sandbox.highMemoryDetail", { total: sandbox.measuredMemory })}
+
{t("sandbox.parked")}{(sandbox.sleeping + sandbox.pending).toLocaleString(locale)}{t("sandbox.parkedDetail", { sleeping: sandbox.sleeping, pending: sandbox.pending })}
+
+ {unavailableReasons.length > 0 || sandbox.highestMemory.length > 0 ? ( +
+
+

{t("sandbox.sampleGaps")}

+ {unavailableReasons.length ?
    {unavailableReasons.map(([reason, count]) =>
  • {t(`runtime.status.${reason}` as never)}{count}
  • )}
:

{t("sandbox.noGaps")}

} +
+
+

{t("sandbox.memoryLeaders")}

+ {sandbox.highestMemory.length ?
    {sandbox.highestMemory.map((entry) =>
  • {entry.memoryPercent.toLocaleString(locale, { maximumFractionDigits: 1 })}%
  • )}
:

{t("sandbox.noMemory")}

} +
+
+ ) : null} +
+
diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx index eeba94ed4..000f8c416 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx @@ -47,7 +47,7 @@ describe("Runtime live-window chart accessibility", () => { })).toBe("Memory usage all series hidden; use the legend to show a series"); }); - it("renders an unavailable current value as zero without retaining a stale value", () => { + it("renders an unavailable current value as a gap without retaining a stale value", () => { const unavailable = { ...sample(120_000, null), activeSandboxCount: 0, @@ -58,13 +58,13 @@ describe("Runtime live-window chart accessibility", () => { , ); - expect(html).toContain("Runtime worker0%0"); + expect(html).toContain("Runtime workerUnavailable1"); expect(html).not.toContain("Runtime worker50%1"); - expect(html).toContain("used0 B0"); + expect(html).toContain("usedUnavailable1"); expect(html).toContain("active00"); }); - it("renders empty retained buckets as continuous zero-value chart series", () => { + it("renders empty retained buckets as missing measurements", () => { const empty = (sampledAt: number): RuntimeTrendSample => ({ ...sample(sampledAt, null), targets: [], @@ -78,13 +78,13 @@ describe("Runtime live-window chart accessibility", () => { , ); - expect(html).toContain("usage0%0"); - expect(html).toContain("used0 B0"); + expect(html).toContain("usageUnavailable2"); + expect(html).toContain("usedUnavailable2"); expect(html).toContain("active00"); - expect(html).toContain("input0/min0"); - expect(html).not.toContain("No retained CPU samples"); - expect(html).not.toContain("No complete retained memory samples"); - expect(html).not.toContain("No retained token samples"); + expect(html).toContain("inputUnavailable2"); + expect(html).toContain("No retained CPU samples"); + expect(html).toContain("No retained observed memory samples"); + expect(html).toContain("No retained token samples"); }); it("renders uPlot chart mounts and reports trends only after two real samples", () => { diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx index c85f06857..d21df864e 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx @@ -537,13 +537,16 @@ export function RuntimeTrendCharts({ id, label: targetLabel(samples, id), tone: tones[index] ?? "blue", - points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: (sample.targets.find((target) => target.seriesId === id)?.cpuRatio ?? 0) * 100 })), + points: samples.map((sample) => { + const ratio = sample.targets.find((target) => target.seriesId === id)?.cpuRatio; + return { sampledAt: sample.sampledAt, value: ratio == null ? null : ratio * 100 }; + }), })); if (cpu.length === 0 && samples.length > 0) { - cpu.push({ id: "cpu", label: t("charts.usage"), tone: "orange", points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: 0 })) }); + cpu.push({ id: "cpu", label: t("charts.usage"), tone: "orange", points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: null })) }); } - const memoryUsed = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryUsageBytes ?? 0 })); - const memoryLimit = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryLimitBytes ?? 0 })); + const memoryUsed = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryUsageBytes })); + const memoryLimit = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryLimitBytes })); const active = [{ id: "active", label: t(activeDisplay === "binary" ? "charts.runtime" : "charts.active.series"), @@ -552,8 +555,8 @@ export function RuntimeTrendCharts({ points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: activeDisplay === "binary" - ? (sample.activeSandboxCount ?? 0) > 0 ? 1 : 0 - : sample.activeSandboxCount ?? 0, + ? sample.activeSandboxCount === null ? null : sample.activeSandboxCount > 0 ? 1 : 0 + : sample.activeSandboxCount, })), }] satisfies TrendSeries[]; const throughput = tokenThroughput(samples); @@ -565,8 +568,8 @@ export function RuntimeTrendCharts({ ] satisfies TrendSeries[], active, tokens: [ - { id: "input", label: t("charts.input"), tone: "orange", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.inputPerMinute ?? 0 })) }, - { id: "output", label: t("charts.output"), tone: "green", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.outputPerMinute ?? 0 })) }, + { id: "input", label: t("charts.input"), tone: "orange", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.inputPerMinute })) }, + { id: "output", label: t("charts.output"), tone: "green", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.outputPerMinute })) }, ] satisfies TrendSeries[], }; }, [activeDisplay, samples, t]); diff --git a/apps/web/src/features/dashboard/dashboard-model.test.ts b/apps/web/src/features/dashboard/dashboard-model.test.ts index 349e6b3ae..8288f1698 100644 --- a/apps/web/src/features/dashboard/dashboard-model.test.ts +++ b/apps/web/src/features/dashboard/dashboard-model.test.ts @@ -5,6 +5,7 @@ import type { AgentSession, RuntimeObservation, SavedAgent, TokenUsage } from "@ import { buildDashboardSnapshot, buildRuntimeDashboardModel, + buildSandboxInsights, dashboardEnvironmentLabel, dashboardEnvironmentProfile, dashboardStatusLabel, @@ -355,6 +356,56 @@ describe("Dashboard loaded-snapshot model", () => { totalTokens: 26, tokenCoverageCount: 2, }); + expect(buildSandboxInsights(model.rows)).toMatchObject({ + observed: 1, + unavailable: 0, + highMemory: 0, + measuredMemory: 1, + highestMemory: [{ sessionId: second.id, memoryPercent: 37.5 }], + }); + }); + + it("separates sleeping allocations from unexpected sampling gaps", () => { + const base = session("11111111-1111-4111-8111-111111111111"); + const observation: RuntimeObservation = { + id: base.id, object: "agent.runtime_observation", session_id: base.id, + environment_id: "33333333-3333-4333-8333-333333333333", mode: "openai_hosted", + provider_type: "docker", + instance: { kind: "managed_allocation", allocation_id: "allocation-1", device_id: null, connection_generation: null }, + lifecycle_state: "sleeping", status: "unavailable", reason: "runtime_not_running", + allocation_created_at: 100, resolved_at: 220, observed_at: null, started_at: null, + cpu: null, memory: null, + }; + const failed = { ...observation, id: "other", session_id: "other", instance: { ...observation.instance, allocation_id: "allocation-2" }, lifecycle_state: "active" as const, reason: "sample_timeout" as const }; + const rows = buildRuntimeDashboardModel([base, session("other")], [observation, failed]).rows; + expect(buildSandboxInsights(rows)).toMatchObject({ + sleeping: 1, pending: 0, unavailable: 1, + unavailableReasons: { sample_timeout: 1 }, + }); + }); + + it("does not pick a conflicting same-second allocation observation by list order", () => { + const first = session("first"); + const second = session("second"); + const base: RuntimeObservation = { + id: first.id, object: "agent.runtime_observation", session_id: first.id, + environment_id: "33333333-3333-4333-8333-333333333333", mode: "openai_hosted", + provider_type: "docker", instance: { kind: "managed_allocation", allocation_id: "allocation-1", device_id: null, connection_generation: null }, + lifecycle_state: "active", status: "observed", reason: null, + allocation_created_at: 100, resolved_at: 220, observed_at: 210, started_at: 150, + cpu: null, memory: { usage_bytes: 900, limit_bytes: 1000 }, + }; + const conflict: RuntimeObservation = { + ...base, id: second.id, session_id: second.id, status: "unavailable", reason: "sample_timeout", + observed_at: null, started_at: null, cpu: null, memory: null, + }; + const insights = (observations: RuntimeObservation[]) => buildSandboxInsights(buildRuntimeDashboardModel([first, second], observations).rows); + for (const order of [[base, conflict], [conflict, base]]) { + expect(insights(order)).toMatchObject({ + observed: 0, unavailable: 1, highMemory: 0, measuredMemory: 0, + unavailableReasons: { sample_unavailable: 1 }, + }); + } }); it("holds each Session's last reported tokens in the summary while public usage is null", () => { diff --git a/apps/web/src/features/dashboard/dashboard-model.ts b/apps/web/src/features/dashboard/dashboard-model.ts index 9f506aef7..7fd967555 100644 --- a/apps/web/src/features/dashboard/dashboard-model.ts +++ b/apps/web/src/features/dashboard/dashboard-model.ts @@ -81,6 +81,26 @@ export interface RuntimeDashboardModel { rows: RuntimeDashboardRow[]; } +export interface SandboxInsight { + allocationId: string; + sessionId: string; + title: string; + memoryPercent: number; + memoryUsageBytes: number; + memoryLimitBytes: number; +} + +export interface SandboxInsights { + observed: number; + unavailable: number; + pending: number; + sleeping: number; + highMemory: number; + measuredMemory: number; + unavailableReasons: Partial, number>>; + highestMemory: SandboxInsight[]; +} + const sessionStatuses = new Set([ "idle", "in_progress", @@ -475,6 +495,75 @@ export function buildRuntimeDashboardModel( }; } +/** Current managed allocations only; duplicate Session observations share one allocation. */ +export function buildSandboxInsights(rows: readonly RuntimeDashboardRow[]): SandboxInsights { + const byAllocation = new Map(); + let unallocatedPending = 0; + for (const row of rows) { + const allocationId = row.observation.instance.allocation_id; + if (row.observation.mode !== "openai_hosted") continue; + if (!allocationId) { + if (row.observation.lifecycle_state === "pending") unallocatedPending += 1; + continue; + } + const previous = byAllocation.get(allocationId); + if (previous) previous.push(row); + else byAllocation.set(allocationId, [row]); + } + const insights: SandboxInsights = { + observed: 0, unavailable: 0, pending: unallocatedPending, sleeping: 0, + highMemory: 0, measuredMemory: 0, unavailableReasons: {}, highestMemory: [], + }; + for (const [allocationId, candidates] of byAllocation) { + const newestTime = candidates.reduce((latest, candidate) => Math.max(latest, candidate.observation.resolved_at), 0); + const newest = candidates.filter((candidate) => candidate.observation.resolved_at === newestTime); + const signature = (candidate: RuntimeDashboardRow) => { + const value = candidate.observation; + return JSON.stringify([value.lifecycle_state, value.status, value.reason, value.observed_at, value.memory]); + }; + // Public timestamps have second resolution. Conflicting observations in one + // second have no reliable order, so do not publish a pressure measurement. + if (new Set(newest.map(signature)).size > 1) { + insights.unavailable += 1; + insights.unavailableReasons.sample_unavailable = (insights.unavailableReasons.sample_unavailable ?? 0) + 1; + continue; + } + const row = newest.sort((left, right) => left.observation.session_id.localeCompare(right.observation.session_id))[0]!; + const observation = row.observation; + if (observation.lifecycle_state === "pending") insights.pending += 1; + if (observation.lifecycle_state === "sleeping") insights.sleeping += 1; + if (observation.status === "unavailable") { + // Parked and stopped allocations are not sampling failures. + if (observation.lifecycle_state !== "sleeping" && observation.lifecycle_state !== "pending" && observation.lifecycle_state !== "stopped") { + insights.unavailable += 1; + if (observation.reason) { + insights.unavailableReasons[observation.reason] = (insights.unavailableReasons[observation.reason] ?? 0) + 1; + } + } + continue; + } + if (observation.status !== "observed") continue; + insights.observed += 1; + const usage = safeNonNegativeInteger(observation.memory?.usage_bytes); + const limit = safeNonNegativeInteger(observation.memory?.limit_bytes); + if (usage === null || limit === null || limit === 0) continue; + insights.measuredMemory += 1; + const memoryPercent = usage / limit * 100; + if (memoryPercent >= 80) insights.highMemory += 1; + insights.highestMemory.push({ + allocationId, + sessionId: row.observation.session_id, + title: row.session.title, + memoryPercent, + memoryUsageBytes: usage, + memoryLimitBytes: limit, + }); + } + insights.highestMemory.sort((left, right) => right.memoryPercent - left.memoryPercent || left.allocationId.localeCompare(right.allocationId)); + insights.highestMemory = insights.highestMemory.slice(0, 5); + return insights; +} + export function formatDashboardBytes(value: number | null): string { if (value === null) return "Unavailable"; const units = ["B", "KiB", "MiB", "GiB", "TiB"]; diff --git a/apps/web/src/i18n/locales/en/dashboard.ts b/apps/web/src/i18n/locales/en/dashboard.ts index 449e5a735..4415511ff 100644 --- a/apps/web/src/i18n/locales/en/dashboard.ts +++ b/apps/web/src/i18n/locales/en/dashboard.ts @@ -1,4 +1,14 @@ export const dashboard = { + sandbox: { + title: "Sandbox diagnostics", subtitle: "Current allocation identity, sampling coverage, and memory pressure", + current: "Current complete snapshot", retained: "Last complete snapshot · refresh failed", + observed: "Observed", observedDetail: "of {{total}} managed targets", + unavailable: "Sampling gaps", unavailableDetail: "Excludes sleeping, pending and stopped", + highMemory: "Memory ≥80%", highMemoryDetail: "of {{total}} with measured usage and limit", + parked: "Parked", parkedDetail: "{{sleeping}} sleeping · {{pending}} pending", + sampleGaps: "Why samples are missing", noGaps: "No unexpected sampling gaps", + memoryLeaders: "Highest memory pressure", noMemory: "No measured memory and limit", + }, runtime: { explorer: "Runtime target explorer", targets: "Runtime targets", resourceSnapshot: "Runtime resource snapshot", search: "Search Runtime targets", searchPlaceholder: "Search Session, provider, or identity", visible: "{{value}} visible", diff --git a/apps/web/src/i18n/locales/zh-CN/dashboard.ts b/apps/web/src/i18n/locales/zh-CN/dashboard.ts index 995f1720b..7c9ca0cf5 100644 --- a/apps/web/src/i18n/locales/zh-CN/dashboard.ts +++ b/apps/web/src/i18n/locales/zh-CN/dashboard.ts @@ -1,4 +1,14 @@ export const dashboard = { + sandbox: { + title: "Sandbox 诊断", subtitle: "按分配身份查看采样覆盖、生命周期与内存压力", + current: "当前完整快照", retained: "上次完整快照 · 刷新失败", + observed: "已观测", observedDetail: "共 {{total}} 个托管目标", + unavailable: "采样缺口", unavailableDetail: "不含休眠、待分配和已停止状态", + highMemory: "内存 ≥80%", highMemoryDetail: "共 {{total}} 个具备用量和上限采样", + parked: "暂停/等待", parkedDetail: "{{sleeping}} 个休眠 · {{pending}} 个等待分配", + sampleGaps: "采样缺失原因", noGaps: "没有意外采样缺口", + memoryLeaders: "内存压力最高", noMemory: "暂无内存用量及上限采样", + }, runtime: { explorer: "Runtime 目标浏览器", targets: "Runtime 目标", resourceSnapshot: "Runtime 资源快照", search: "搜索 Runtime 目标", searchPlaceholder: "搜索会话、Provider 或标识", visible: "显示 {{value}} 项", diff --git a/contracts/agents-api/runtime-observability.md b/contracts/agents-api/runtime-observability.md index 143c45b3b..1962abb51 100644 --- a/contracts/agents-api/runtime-observability.md +++ b/contracts/agents-api/runtime-observability.md @@ -83,9 +83,20 @@ Web projects active Runtime state differently by scope. The Dashboard shows one summed series of distinct allocation identities: live snapshots count `lifecycle_state: active`, while retained buckets count successfully observed allocations because lifecycle state is not retained yet. The single-Session view -collapses the same value to `1` or `0`. Missing or unavailable retained values are -currently rendered as zero, so this presentation intentionally does not yet -distinguish sleeping from collection failure. +collapses the same value to `1` or `0`. Empty retained buckets have no active +observation; an observed zero in a current lifecycle snapshot remains zero. +Retained history does not distinguish sleeping from collection failure because it +does not retain lifecycle state. + +The Dashboard's current Sandbox diagnostics deduplicate managed observations by +allocation ID, report sampling gaps separately from sleeping, pending and stopped +allocations, and rank measured memory usage against a known limit. A target with +missing usage or limit has no pressure percentage. Live and retained chart gaps +remain null rather than fabricated zero; an observed lifecycle count of zero is +still zero. Conflicting observations of one allocation with the same public +second-resolution timestamp count as an ambiguous sample gap. The broader +request, model and tool collection plan is documented in +[system observability](system-observability-plan.md). Future automatic suspension requires a separate durable control model, including an activity revision and timestamps such as `idle_since` and diff --git a/contracts/agents-api/system-observability-plan.md b/contracts/agents-api/system-observability-plan.md new file mode 100644 index 000000000..1f59827c4 --- /dev/null +++ b/contracts/agents-api/system-observability-plan.md @@ -0,0 +1,79 @@ +# System observability collection plan + +Status: collection inventory and next-phase design. The current Dashboard release +uses existing Runtime observations and Session Usage only. Request, model, and tool +panels must not be shown as measured until the instruments below are implemented. + +## Existing evidence + +| Signal | Source | Current coverage | Dashboard meaning | +| --- | --- | --- | --- | +| Sandbox lifecycle | Core allocation state | Managed Docker and microsandbox | Current control state, not execution success | +| CPU time and capacity | Provider observation | Current and periodic history when sampling is available | Cumulative time; utilization needs two samples from one compute incarnation | +| Memory usage and limit | Provider observation | Current and periodic history when reported | Point-in-time guest/container memory, not host memory | +| Token input/output | Measured Session Usage | Public current Session and periodic Runtime history | Reported canonical usage; absent usage remains unknown | +| Request count, latency, errors | No common metric | Not available | Do not derive from Session count or HTTP page loads | +| Model distribution and latency | Saved Agent model is available, terminal call records are not aggregated | Configuration only | Do not label configured models as invoked models | +| Tool calls and failures | Durable Turn events exist, no bounded aggregate read | Partial event evidence | Do not count tools by Agent declarations | +| Host, database, queue health | No tenant-safe operator projection | Not available | Keep out of Session-scoped charts | + +Current Runtime history has a 30-second default periodic collection interval and +seven-day retention; the public Web ranges are 1h, 6h, and 24h. Its capability +route distinguishes periodic collection from on-read collection. The Dashboard +must show the source, age, coverage and missing values. An empty interval is not +an observed zero. + +## Next collection boundary + +Add a separate, authenticated operator metrics service behind Core. Keep the +pinned Agents API resources unchanged. The service should aggregate sanitized +events server-side and expose a bounded read-only extension through +`packages/agents-client`. Browser input may select a fixed range and resolution; +it must not select tenants, storage labels, arbitrary PromQL, endpoints or keys. + +1. **Ingress:** count completed HTTP requests by route family, method and coarse + outcome (`success`, `client_error`, `server_error`). Record a latency histogram + after the response completes. Exclude health polling or show it separately. + This is transport health, never terminal Turn success. +2. **Execution:** emit one terminal Turn outcome from the durable state transition + winner, with queue wait and run duration measured from persisted timestamps. + Active Turns are a current gauge from Core ownership, not a counter inferred + from sampled CPU. Retries and recovery must not double count terminal outcomes. +3. **Model:** observe actual native model invocation attempts at the harness + boundary, including model identifier, terminal attempt outcome, duration, + reported input/output tokens and usage coverage. A saved Agent's configured + model is not proof of which model ran. Keep bounded model label cardinality. +4. **Tools:** observe actual tool attempt start and terminal result at the common + Runtime adapter boundary. Use a bounded tool category and outcome label; + raw tool names, arguments, output and credentials stay out of metric labels. + Distinguish model tool selection from completed tool execution. +5. **Sandbox:** retain allocation identity and compute generation internally. + Extend the normalized sample only after Docker and microsandbox values have + matching semantics. Candidates are OOM/exit events, disk usage/limit, network + bytes, compute restarts, and allocation/compute startup latency. A missing + provider value remains null; lifecycle state remains Core-owned. +6. **Collector health:** count attempted, successful, timed-out and unavailable + samples, queue drops and export failures. Display coverage per interval so + a quiet chart cannot hide collector failure. + +Use bounded histograms for p50/p95 latency and rates from counters over complete +time buckets. Keep operational cardinality to route family, outcome, provider +type and bounded model family. Session, allocation, tenant, tool name and native +identifiers must not become general metrics labels. Drill-down can use authorized +Core resource IDs through existing tenant-scoped reads. Retention and query limits +must be explicit, with history storage/export optional for execution. + +## Dashboard layout and acceptance + +The overview keeps loaded Agent/Session counts and current attention. The Runtime +section shows allocation-deduplicated lifecycle, coverage, sample gaps, memory +pressure and existing Live/History trends. A later System section can add request +rate/error rate/p95, terminal Turn outcomes and durations, actual model usage and +tool attempts once their collection is qualified. Every panel needs a source and +freshness label, a coverage denominator, and an unavailable state. It must never +convert missing measurements to zero or equate a configured Sandbox with a +completed execution. + +Before shipping those new panels, validate counter deduplication under retries, +provider restarts, missing usage, sampler outages and multi-tenant authorization; +verify that instrumentation has no effect on dispatch, lifecycle or settlement.