From bd5972962d11e523ba726b6e2a206bdd6ce28412 Mon Sep 17 00:00:00 2001 From: sam Date: Fri, 25 Sep 2026 01:27:40 +0800 Subject: [PATCH 1/2] Add allocation-aware Sandbox diagnostics to Dashboard --- apps/web/e2e/agents-lifecycle.spec.ts | 11 ++- .../src/features/dashboard/DashboardView.css | 58 ++++++++++++ .../features/dashboard/DashboardView.test.tsx | 6 ++ .../dashboard/RuntimeObservabilityContent.tsx | 32 +++++++ .../dashboard/RuntimeTrendCharts.test.tsx | 20 ++--- .../features/dashboard/RuntimeTrendCharts.tsx | 19 ++-- .../dashboard/dashboard-model.test.ts | 51 +++++++++++ .../src/features/dashboard/dashboard-model.ts | 89 +++++++++++++++++++ apps/web/src/i18n/locales/en/dashboard.ts | 10 +++ apps/web/src/i18n/locales/zh-CN/dashboard.ts | 10 +++ contracts/agents-api/runtime-observability.md | 17 +++- .../agents-api/system-observability-plan.md | 79 ++++++++++++++++ 12 files changed, 378 insertions(+), 24 deletions(-) create mode 100644 contracts/agents-api/system-observability-plan.md diff --git a/apps/web/e2e/agents-lifecycle.spec.ts b/apps/web/e2e/agents-lifecycle.spec.ts index 9de471d89..7b63cb271 100644 --- a/apps/web/e2e/agents-lifecycle.spec.ts +++ b/apps/web/e2e/agents-lifecycle.spec.ts @@ -2684,6 +2684,11 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand const dashboard = page.locator(".dashboard-page"); await expect(dashboard.getByRole("heading", { name: "Dashboard", exact: true })).toBeVisible(); const refresh = dashboard.getByRole("button", { name: "Refresh Dashboard snapshot" }); + const sandboxDiagnostics = dashboard.getByRole("region", { name: "Sandbox diagnostics" }); + await expect(sandboxDiagnostics).toBeVisible(); + await expect(sandboxDiagnostics).toContainText("12 managed targets"); + await expect(sandboxDiagnostics).toContainText("12 with measured usage and limit"); + await expect(sandboxDiagnostics).toContainText("Highest memory pressure"); await expect(dashboard.locator(".dashboard-runtime-sample-count")).toContainText("1 sample ·"); await expect(dashboard.getByRole("heading", { name: "CPU usage" })).toBeVisible(); await expect(dashboard.getByRole("heading", { name: "Memory usage" })).toBeVisible(); @@ -2735,7 +2740,7 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand await expect(cpuCard.locator(".dashboard-runtime-trend-tooltip")).toBeVisible(); await cpuChart.click({ position: { x: 260, y: 90 } }); await expect(cpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("Pinned"); - await expect(cpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("10%"); + await expect(cpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("Unavailable"); await cpuChart.focus(); await cpuChart.press("ArrowRight"); await expect(cpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("Pinned"); @@ -2930,9 +2935,9 @@ test("restores retained Runtime history after a Dashboard reload", async ({ page const durableCpuCard = durableCpuChart.locator("xpath=ancestor::section[contains(@class, 'dashboard-runtime-trend-card')]"); await durableCpuChart.focus(); await durableCpuChart.press("ArrowLeft"); - await expect(durableCpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("0%"); + await expect(durableCpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("Unavailable"); await durableCpuChart.press("ArrowLeft"); - await expect(durableCpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("0%"); + await expect(durableCpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("Unavailable"); const durableMemoryCard = dashboard.getByRole("region", { name: "Memory usage durable history chart" }); const durableMemorySpan = await durableMemoryCard.locator("canvas").evaluate((canvas: HTMLCanvasElement) => { const context = canvas.getContext("2d"); diff --git a/apps/web/src/features/dashboard/DashboardView.css b/apps/web/src/features/dashboard/DashboardView.css index 994880b84..5364e670c 100644 --- a/apps/web/src/features/dashboard/DashboardView.css +++ b/apps/web/src/features/dashboard/DashboardView.css @@ -633,6 +633,64 @@ border-bottom: 1px solid var(--line); } +.dashboard-sandbox-insights { + padding: 16px; + border-bottom: 1px solid var(--line); +} + +.dashboard-sandbox-insights > header, +.dashboard-sandbox-diagnostics { + display: flex; + justify-content: space-between; + gap: 16px; +} + +.dashboard-sandbox-insights h3, +.dashboard-sandbox-insights h4, +.dashboard-sandbox-insights p { + margin: 0; +} + +.dashboard-sandbox-insights h3 { font-size: 14px; } +.dashboard-sandbox-insights h4 { font-size: 12px; } +.dashboard-sandbox-insights p, +.dashboard-sandbox-insights small { color: var(--fg-muted); font-size: 11px; } + +.dashboard-sandbox-insight-grid { + display: grid; + grid-template-columns: repeat(4, minmax(0, 1fr)); + gap: 8px; + margin-top: 12px; +} + +.dashboard-sandbox-insight-grid > div { + display: grid; + gap: 3px; + padding: 10px 12px; + border: 1px solid var(--line); + border-radius: 8px; + min-width: 0; +} + +.dashboard-sandbox-insight-grid strong { + font: 550 20px/1.2 var(--font-mono); + font-variant-numeric: tabular-nums; +} + +.dashboard-sandbox-insight-grid span { color: var(--fg-muted); font-size: 10px; } +.dashboard-sandbox-diagnostics { margin-top: 14px; } +.dashboard-sandbox-diagnostics > div { flex: 1; min-width: 0; } +.dashboard-sandbox-diagnostics ul { list-style: none; margin: 7px 0 0; padding: 0; } +.dashboard-sandbox-diagnostics li { display: flex; justify-content: space-between; gap: 8px; padding: 4px 0; font-size: 11px; } +.dashboard-sandbox-diagnostics li + li { border-top: 1px solid var(--line); } +.dashboard-sandbox-diagnostics li button { padding: 0; border: 0; background: none; color: var(--accent); cursor: pointer; text-align: left; } +.dashboard-sandbox-diagnostics li span { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } + +@media (max-width: 800px) { + .dashboard-sandbox-insight-grid { grid-template-columns: repeat(2, minmax(0, 1fr)); } + .dashboard-sandbox-diagnostics { flex-direction: column; } +} + .dashboard-runtime-metric { display: grid; min-width: 0; diff --git a/apps/web/src/features/dashboard/DashboardView.test.tsx b/apps/web/src/features/dashboard/DashboardView.test.tsx index 39f3a23e0..4ad132758 100644 --- a/apps/web/src/features/dashboard/DashboardView.test.tsx +++ b/apps/web/src/features/dashboard/DashboardView.test.tsx @@ -247,6 +247,12 @@ describe("Dashboard loaded-result presentation", () => { expect(html).toContain("Cumulative CPU / capacity"); expect(html).toContain("1m 13s / 2 cores"); expect(html).toContain("512 MiB / 2.00 GiB"); + expect(html).toContain("Sandbox diagnostics"); + expect(html).toContain("Memory ≥80%"); + const withoutMemory = render({ + runtimeSnapshot: { sessions: [hosted], observations: [{ ...observation, memory: { usage_bytes: null, limit_bytes: null } }], loadedAt: 1_700_000_100_000 }, + }); + expect(withoutMemory).toMatch(/Memory ≥80%<\/small>Unavailable<\/strong>/); expect(html).toContain('aria-label="Runtime live-window charts"'); expect(html).toContain("Resource trends"); expect(html).toContain("Browser-local samples · reset on reload"); diff --git a/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx b/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx index 2c39d9d37..f40f5e62e 100644 --- a/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx +++ b/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx @@ -26,6 +26,7 @@ import { import { buildRuntimeDashboardModel, + buildSandboxInsights, formatDashboardBytes, formatDashboardDuration, formatDashboardTimestamp, @@ -312,6 +313,9 @@ export function RuntimeObservabilityContent({ [snapshot, tokenTotals], ); const summary = model.summary; + const sandbox = useMemo(() => buildSandboxInsights(model.rows), [model.rows]); + const unavailableReasons = Object.entries(sandbox.unavailableReasons) + .sort((left, right) => right[1] - left[1]); return ( <>
@@ -321,6 +325,34 @@ export function RuntimeObservabilityContent({ } label={t("runtime.metrics.tokens")} value={summary.totalTokens === null ? t("runtime.filters.unavailable") : formatDashboardTokens(summary.totalTokens, locale)} detail={t("runtime.metrics.tokenDetail", { covered: summary.tokenCoverageCount, total: summary.sessionCount })} />
+
+
+
+

{t("sandbox.title")}

+

{t("sandbox.subtitle")}

+
+ {t(stale ? "sandbox.retained" : "sandbox.current")} +
+
+
{t("sandbox.observed")}{sandbox.observed.toLocaleString(locale)}{t("sandbox.observedDetail", { total: summary.managedRuntimeCount })}
+
{t("sandbox.unavailable")}{sandbox.unavailable.toLocaleString(locale)}{t("sandbox.unavailableDetail")}
+
{t("sandbox.highMemory")}{sandbox.measuredMemory === 0 ? t("runtime.filters.unavailable") : sandbox.highMemory.toLocaleString(locale)}{t("sandbox.highMemoryDetail", { total: sandbox.measuredMemory })}
+
{t("sandbox.parked")}{(sandbox.sleeping + sandbox.pending).toLocaleString(locale)}{t("sandbox.parkedDetail", { sleeping: sandbox.sleeping, pending: sandbox.pending })}
+
+ {unavailableReasons.length > 0 || sandbox.highestMemory.length > 0 ? ( +
+
+

{t("sandbox.sampleGaps")}

+ {unavailableReasons.length ?
    {unavailableReasons.map(([reason, count]) =>
  • {t(`runtime.status.${reason}` as never)}{count}
  • )}
:

{t("sandbox.noGaps")}

} +
+
+

{t("sandbox.memoryLeaders")}

+ {sandbox.highestMemory.length ?
    {sandbox.highestMemory.map((entry) =>
  • {entry.memoryPercent.toLocaleString(locale, { maximumFractionDigits: 1 })}%
  • )}
:

{t("sandbox.noMemory")}

} +
+
+ ) : null} +
+
diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx index eeba94ed4..000f8c416 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx @@ -47,7 +47,7 @@ describe("Runtime live-window chart accessibility", () => { })).toBe("Memory usage all series hidden; use the legend to show a series"); }); - it("renders an unavailable current value as zero without retaining a stale value", () => { + it("renders an unavailable current value as a gap without retaining a stale value", () => { const unavailable = { ...sample(120_000, null), activeSandboxCount: 0, @@ -58,13 +58,13 @@ describe("Runtime live-window chart accessibility", () => { , ); - expect(html).toContain("Runtime worker0%0"); + expect(html).toContain("Runtime workerUnavailable1"); expect(html).not.toContain("Runtime worker50%1"); - expect(html).toContain("used0 B0"); + expect(html).toContain("usedUnavailable1"); expect(html).toContain("active00"); }); - it("renders empty retained buckets as continuous zero-value chart series", () => { + it("renders empty retained buckets as missing measurements", () => { const empty = (sampledAt: number): RuntimeTrendSample => ({ ...sample(sampledAt, null), targets: [], @@ -78,13 +78,13 @@ describe("Runtime live-window chart accessibility", () => { , ); - expect(html).toContain("usage0%0"); - expect(html).toContain("used0 B0"); + expect(html).toContain("usageUnavailable2"); + expect(html).toContain("usedUnavailable2"); expect(html).toContain("active00"); - expect(html).toContain("input0/min0"); - expect(html).not.toContain("No retained CPU samples"); - expect(html).not.toContain("No complete retained memory samples"); - expect(html).not.toContain("No retained token samples"); + expect(html).toContain("inputUnavailable2"); + expect(html).toContain("No retained CPU samples"); + expect(html).toContain("No retained observed memory samples"); + expect(html).toContain("No retained token samples"); }); it("renders uPlot chart mounts and reports trends only after two real samples", () => { diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx index c85f06857..d21df864e 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx @@ -537,13 +537,16 @@ export function RuntimeTrendCharts({ id, label: targetLabel(samples, id), tone: tones[index] ?? "blue", - points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: (sample.targets.find((target) => target.seriesId === id)?.cpuRatio ?? 0) * 100 })), + points: samples.map((sample) => { + const ratio = sample.targets.find((target) => target.seriesId === id)?.cpuRatio; + return { sampledAt: sample.sampledAt, value: ratio == null ? null : ratio * 100 }; + }), })); if (cpu.length === 0 && samples.length > 0) { - cpu.push({ id: "cpu", label: t("charts.usage"), tone: "orange", points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: 0 })) }); + cpu.push({ id: "cpu", label: t("charts.usage"), tone: "orange", points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: null })) }); } - const memoryUsed = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryUsageBytes ?? 0 })); - const memoryLimit = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryLimitBytes ?? 0 })); + const memoryUsed = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryUsageBytes })); + const memoryLimit = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryLimitBytes })); const active = [{ id: "active", label: t(activeDisplay === "binary" ? "charts.runtime" : "charts.active.series"), @@ -552,8 +555,8 @@ export function RuntimeTrendCharts({ points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: activeDisplay === "binary" - ? (sample.activeSandboxCount ?? 0) > 0 ? 1 : 0 - : sample.activeSandboxCount ?? 0, + ? sample.activeSandboxCount === null ? null : sample.activeSandboxCount > 0 ? 1 : 0 + : sample.activeSandboxCount, })), }] satisfies TrendSeries[]; const throughput = tokenThroughput(samples); @@ -565,8 +568,8 @@ export function RuntimeTrendCharts({ ] satisfies TrendSeries[], active, tokens: [ - { id: "input", label: t("charts.input"), tone: "orange", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.inputPerMinute ?? 0 })) }, - { id: "output", label: t("charts.output"), tone: "green", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.outputPerMinute ?? 0 })) }, + { id: "input", label: t("charts.input"), tone: "orange", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.inputPerMinute })) }, + { id: "output", label: t("charts.output"), tone: "green", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.outputPerMinute })) }, ] satisfies TrendSeries[], }; }, [activeDisplay, samples, t]); diff --git a/apps/web/src/features/dashboard/dashboard-model.test.ts b/apps/web/src/features/dashboard/dashboard-model.test.ts index 349e6b3ae..8288f1698 100644 --- a/apps/web/src/features/dashboard/dashboard-model.test.ts +++ b/apps/web/src/features/dashboard/dashboard-model.test.ts @@ -5,6 +5,7 @@ import type { AgentSession, RuntimeObservation, SavedAgent, TokenUsage } from "@ import { buildDashboardSnapshot, buildRuntimeDashboardModel, + buildSandboxInsights, dashboardEnvironmentLabel, dashboardEnvironmentProfile, dashboardStatusLabel, @@ -355,6 +356,56 @@ describe("Dashboard loaded-snapshot model", () => { totalTokens: 26, tokenCoverageCount: 2, }); + expect(buildSandboxInsights(model.rows)).toMatchObject({ + observed: 1, + unavailable: 0, + highMemory: 0, + measuredMemory: 1, + highestMemory: [{ sessionId: second.id, memoryPercent: 37.5 }], + }); + }); + + it("separates sleeping allocations from unexpected sampling gaps", () => { + const base = session("11111111-1111-4111-8111-111111111111"); + const observation: RuntimeObservation = { + id: base.id, object: "agent.runtime_observation", session_id: base.id, + environment_id: "33333333-3333-4333-8333-333333333333", mode: "openai_hosted", + provider_type: "docker", + instance: { kind: "managed_allocation", allocation_id: "allocation-1", device_id: null, connection_generation: null }, + lifecycle_state: "sleeping", status: "unavailable", reason: "runtime_not_running", + allocation_created_at: 100, resolved_at: 220, observed_at: null, started_at: null, + cpu: null, memory: null, + }; + const failed = { ...observation, id: "other", session_id: "other", instance: { ...observation.instance, allocation_id: "allocation-2" }, lifecycle_state: "active" as const, reason: "sample_timeout" as const }; + const rows = buildRuntimeDashboardModel([base, session("other")], [observation, failed]).rows; + expect(buildSandboxInsights(rows)).toMatchObject({ + sleeping: 1, pending: 0, unavailable: 1, + unavailableReasons: { sample_timeout: 1 }, + }); + }); + + it("does not pick a conflicting same-second allocation observation by list order", () => { + const first = session("first"); + const second = session("second"); + const base: RuntimeObservation = { + id: first.id, object: "agent.runtime_observation", session_id: first.id, + environment_id: "33333333-3333-4333-8333-333333333333", mode: "openai_hosted", + provider_type: "docker", instance: { kind: "managed_allocation", allocation_id: "allocation-1", device_id: null, connection_generation: null }, + lifecycle_state: "active", status: "observed", reason: null, + allocation_created_at: 100, resolved_at: 220, observed_at: 210, started_at: 150, + cpu: null, memory: { usage_bytes: 900, limit_bytes: 1000 }, + }; + const conflict: RuntimeObservation = { + ...base, id: second.id, session_id: second.id, status: "unavailable", reason: "sample_timeout", + observed_at: null, started_at: null, cpu: null, memory: null, + }; + const insights = (observations: RuntimeObservation[]) => buildSandboxInsights(buildRuntimeDashboardModel([first, second], observations).rows); + for (const order of [[base, conflict], [conflict, base]]) { + expect(insights(order)).toMatchObject({ + observed: 0, unavailable: 1, highMemory: 0, measuredMemory: 0, + unavailableReasons: { sample_unavailable: 1 }, + }); + } }); it("holds each Session's last reported tokens in the summary while public usage is null", () => { diff --git a/apps/web/src/features/dashboard/dashboard-model.ts b/apps/web/src/features/dashboard/dashboard-model.ts index 9f506aef7..7fd967555 100644 --- a/apps/web/src/features/dashboard/dashboard-model.ts +++ b/apps/web/src/features/dashboard/dashboard-model.ts @@ -81,6 +81,26 @@ export interface RuntimeDashboardModel { rows: RuntimeDashboardRow[]; } +export interface SandboxInsight { + allocationId: string; + sessionId: string; + title: string; + memoryPercent: number; + memoryUsageBytes: number; + memoryLimitBytes: number; +} + +export interface SandboxInsights { + observed: number; + unavailable: number; + pending: number; + sleeping: number; + highMemory: number; + measuredMemory: number; + unavailableReasons: Partial, number>>; + highestMemory: SandboxInsight[]; +} + const sessionStatuses = new Set([ "idle", "in_progress", @@ -475,6 +495,75 @@ export function buildRuntimeDashboardModel( }; } +/** Current managed allocations only; duplicate Session observations share one allocation. */ +export function buildSandboxInsights(rows: readonly RuntimeDashboardRow[]): SandboxInsights { + const byAllocation = new Map(); + let unallocatedPending = 0; + for (const row of rows) { + const allocationId = row.observation.instance.allocation_id; + if (row.observation.mode !== "openai_hosted") continue; + if (!allocationId) { + if (row.observation.lifecycle_state === "pending") unallocatedPending += 1; + continue; + } + const previous = byAllocation.get(allocationId); + if (previous) previous.push(row); + else byAllocation.set(allocationId, [row]); + } + const insights: SandboxInsights = { + observed: 0, unavailable: 0, pending: unallocatedPending, sleeping: 0, + highMemory: 0, measuredMemory: 0, unavailableReasons: {}, highestMemory: [], + }; + for (const [allocationId, candidates] of byAllocation) { + const newestTime = candidates.reduce((latest, candidate) => Math.max(latest, candidate.observation.resolved_at), 0); + const newest = candidates.filter((candidate) => candidate.observation.resolved_at === newestTime); + const signature = (candidate: RuntimeDashboardRow) => { + const value = candidate.observation; + return JSON.stringify([value.lifecycle_state, value.status, value.reason, value.observed_at, value.memory]); + }; + // Public timestamps have second resolution. Conflicting observations in one + // second have no reliable order, so do not publish a pressure measurement. + if (new Set(newest.map(signature)).size > 1) { + insights.unavailable += 1; + insights.unavailableReasons.sample_unavailable = (insights.unavailableReasons.sample_unavailable ?? 0) + 1; + continue; + } + const row = newest.sort((left, right) => left.observation.session_id.localeCompare(right.observation.session_id))[0]!; + const observation = row.observation; + if (observation.lifecycle_state === "pending") insights.pending += 1; + if (observation.lifecycle_state === "sleeping") insights.sleeping += 1; + if (observation.status === "unavailable") { + // Parked and stopped allocations are not sampling failures. + if (observation.lifecycle_state !== "sleeping" && observation.lifecycle_state !== "pending" && observation.lifecycle_state !== "stopped") { + insights.unavailable += 1; + if (observation.reason) { + insights.unavailableReasons[observation.reason] = (insights.unavailableReasons[observation.reason] ?? 0) + 1; + } + } + continue; + } + if (observation.status !== "observed") continue; + insights.observed += 1; + const usage = safeNonNegativeInteger(observation.memory?.usage_bytes); + const limit = safeNonNegativeInteger(observation.memory?.limit_bytes); + if (usage === null || limit === null || limit === 0) continue; + insights.measuredMemory += 1; + const memoryPercent = usage / limit * 100; + if (memoryPercent >= 80) insights.highMemory += 1; + insights.highestMemory.push({ + allocationId, + sessionId: row.observation.session_id, + title: row.session.title, + memoryPercent, + memoryUsageBytes: usage, + memoryLimitBytes: limit, + }); + } + insights.highestMemory.sort((left, right) => right.memoryPercent - left.memoryPercent || left.allocationId.localeCompare(right.allocationId)); + insights.highestMemory = insights.highestMemory.slice(0, 5); + return insights; +} + export function formatDashboardBytes(value: number | null): string { if (value === null) return "Unavailable"; const units = ["B", "KiB", "MiB", "GiB", "TiB"]; diff --git a/apps/web/src/i18n/locales/en/dashboard.ts b/apps/web/src/i18n/locales/en/dashboard.ts index 449e5a735..4415511ff 100644 --- a/apps/web/src/i18n/locales/en/dashboard.ts +++ b/apps/web/src/i18n/locales/en/dashboard.ts @@ -1,4 +1,14 @@ export const dashboard = { + sandbox: { + title: "Sandbox diagnostics", subtitle: "Current allocation identity, sampling coverage, and memory pressure", + current: "Current complete snapshot", retained: "Last complete snapshot · refresh failed", + observed: "Observed", observedDetail: "of {{total}} managed targets", + unavailable: "Sampling gaps", unavailableDetail: "Excludes sleeping, pending and stopped", + highMemory: "Memory ≥80%", highMemoryDetail: "of {{total}} with measured usage and limit", + parked: "Parked", parkedDetail: "{{sleeping}} sleeping · {{pending}} pending", + sampleGaps: "Why samples are missing", noGaps: "No unexpected sampling gaps", + memoryLeaders: "Highest memory pressure", noMemory: "No measured memory and limit", + }, runtime: { explorer: "Runtime target explorer", targets: "Runtime targets", resourceSnapshot: "Runtime resource snapshot", search: "Search Runtime targets", searchPlaceholder: "Search Session, provider, or identity", visible: "{{value}} visible", diff --git a/apps/web/src/i18n/locales/zh-CN/dashboard.ts b/apps/web/src/i18n/locales/zh-CN/dashboard.ts index 995f1720b..7c9ca0cf5 100644 --- a/apps/web/src/i18n/locales/zh-CN/dashboard.ts +++ b/apps/web/src/i18n/locales/zh-CN/dashboard.ts @@ -1,4 +1,14 @@ export const dashboard = { + sandbox: { + title: "Sandbox 诊断", subtitle: "按分配身份查看采样覆盖、生命周期与内存压力", + current: "当前完整快照", retained: "上次完整快照 · 刷新失败", + observed: "已观测", observedDetail: "共 {{total}} 个托管目标", + unavailable: "采样缺口", unavailableDetail: "不含休眠、待分配和已停止状态", + highMemory: "内存 ≥80%", highMemoryDetail: "共 {{total}} 个具备用量和上限采样", + parked: "暂停/等待", parkedDetail: "{{sleeping}} 个休眠 · {{pending}} 个等待分配", + sampleGaps: "采样缺失原因", noGaps: "没有意外采样缺口", + memoryLeaders: "内存压力最高", noMemory: "暂无内存用量及上限采样", + }, runtime: { explorer: "Runtime 目标浏览器", targets: "Runtime 目标", resourceSnapshot: "Runtime 资源快照", search: "搜索 Runtime 目标", searchPlaceholder: "搜索会话、Provider 或标识", visible: "显示 {{value}} 项", diff --git a/contracts/agents-api/runtime-observability.md b/contracts/agents-api/runtime-observability.md index 143c45b3b..1962abb51 100644 --- a/contracts/agents-api/runtime-observability.md +++ b/contracts/agents-api/runtime-observability.md @@ -83,9 +83,20 @@ Web projects active Runtime state differently by scope. The Dashboard shows one summed series of distinct allocation identities: live snapshots count `lifecycle_state: active`, while retained buckets count successfully observed allocations because lifecycle state is not retained yet. The single-Session view -collapses the same value to `1` or `0`. Missing or unavailable retained values are -currently rendered as zero, so this presentation intentionally does not yet -distinguish sleeping from collection failure. +collapses the same value to `1` or `0`. Empty retained buckets have no active +observation; an observed zero in a current lifecycle snapshot remains zero. +Retained history does not distinguish sleeping from collection failure because it +does not retain lifecycle state. + +The Dashboard's current Sandbox diagnostics deduplicate managed observations by +allocation ID, report sampling gaps separately from sleeping, pending and stopped +allocations, and rank measured memory usage against a known limit. A target with +missing usage or limit has no pressure percentage. Live and retained chart gaps +remain null rather than fabricated zero; an observed lifecycle count of zero is +still zero. Conflicting observations of one allocation with the same public +second-resolution timestamp count as an ambiguous sample gap. The broader +request, model and tool collection plan is documented in +[system observability](system-observability-plan.md). Future automatic suspension requires a separate durable control model, including an activity revision and timestamps such as `idle_since` and diff --git a/contracts/agents-api/system-observability-plan.md b/contracts/agents-api/system-observability-plan.md new file mode 100644 index 000000000..1f59827c4 --- /dev/null +++ b/contracts/agents-api/system-observability-plan.md @@ -0,0 +1,79 @@ +# System observability collection plan + +Status: collection inventory and next-phase design. The current Dashboard release +uses existing Runtime observations and Session Usage only. Request, model, and tool +panels must not be shown as measured until the instruments below are implemented. + +## Existing evidence + +| Signal | Source | Current coverage | Dashboard meaning | +| --- | --- | --- | --- | +| Sandbox lifecycle | Core allocation state | Managed Docker and microsandbox | Current control state, not execution success | +| CPU time and capacity | Provider observation | Current and periodic history when sampling is available | Cumulative time; utilization needs two samples from one compute incarnation | +| Memory usage and limit | Provider observation | Current and periodic history when reported | Point-in-time guest/container memory, not host memory | +| Token input/output | Measured Session Usage | Public current Session and periodic Runtime history | Reported canonical usage; absent usage remains unknown | +| Request count, latency, errors | No common metric | Not available | Do not derive from Session count or HTTP page loads | +| Model distribution and latency | Saved Agent model is available, terminal call records are not aggregated | Configuration only | Do not label configured models as invoked models | +| Tool calls and failures | Durable Turn events exist, no bounded aggregate read | Partial event evidence | Do not count tools by Agent declarations | +| Host, database, queue health | No tenant-safe operator projection | Not available | Keep out of Session-scoped charts | + +Current Runtime history has a 30-second default periodic collection interval and +seven-day retention; the public Web ranges are 1h, 6h, and 24h. Its capability +route distinguishes periodic collection from on-read collection. The Dashboard +must show the source, age, coverage and missing values. An empty interval is not +an observed zero. + +## Next collection boundary + +Add a separate, authenticated operator metrics service behind Core. Keep the +pinned Agents API resources unchanged. The service should aggregate sanitized +events server-side and expose a bounded read-only extension through +`packages/agents-client`. Browser input may select a fixed range and resolution; +it must not select tenants, storage labels, arbitrary PromQL, endpoints or keys. + +1. **Ingress:** count completed HTTP requests by route family, method and coarse + outcome (`success`, `client_error`, `server_error`). Record a latency histogram + after the response completes. Exclude health polling or show it separately. + This is transport health, never terminal Turn success. +2. **Execution:** emit one terminal Turn outcome from the durable state transition + winner, with queue wait and run duration measured from persisted timestamps. + Active Turns are a current gauge from Core ownership, not a counter inferred + from sampled CPU. Retries and recovery must not double count terminal outcomes. +3. **Model:** observe actual native model invocation attempts at the harness + boundary, including model identifier, terminal attempt outcome, duration, + reported input/output tokens and usage coverage. A saved Agent's configured + model is not proof of which model ran. Keep bounded model label cardinality. +4. **Tools:** observe actual tool attempt start and terminal result at the common + Runtime adapter boundary. Use a bounded tool category and outcome label; + raw tool names, arguments, output and credentials stay out of metric labels. + Distinguish model tool selection from completed tool execution. +5. **Sandbox:** retain allocation identity and compute generation internally. + Extend the normalized sample only after Docker and microsandbox values have + matching semantics. Candidates are OOM/exit events, disk usage/limit, network + bytes, compute restarts, and allocation/compute startup latency. A missing + provider value remains null; lifecycle state remains Core-owned. +6. **Collector health:** count attempted, successful, timed-out and unavailable + samples, queue drops and export failures. Display coverage per interval so + a quiet chart cannot hide collector failure. + +Use bounded histograms for p50/p95 latency and rates from counters over complete +time buckets. Keep operational cardinality to route family, outcome, provider +type and bounded model family. Session, allocation, tenant, tool name and native +identifiers must not become general metrics labels. Drill-down can use authorized +Core resource IDs through existing tenant-scoped reads. Retention and query limits +must be explicit, with history storage/export optional for execution. + +## Dashboard layout and acceptance + +The overview keeps loaded Agent/Session counts and current attention. The Runtime +section shows allocation-deduplicated lifecycle, coverage, sample gaps, memory +pressure and existing Live/History trends. A later System section can add request +rate/error rate/p95, terminal Turn outcomes and durations, actual model usage and +tool attempts once their collection is qualified. Every panel needs a source and +freshness label, a coverage denominator, and an unavailable state. It must never +convert missing measurements to zero or equate a configured Sandbox with a +completed execution. + +Before shipping those new panels, validate counter deduplication under retries, +provider restarts, missing usage, sampler outages and multi-tenant authorization; +verify that instrumentation has no effect on dispatch, lifecycle or settlement. From c4bea994c52da4a9740a24e1e593efb60bf9289f Mon Sep 17 00:00:00 2001 From: sam Date: Fri, 25 Sep 2026 12:28:39 +0800 Subject: [PATCH 2/2] Build operator observability dashboard and telemetry --- CONTRIBUTING.md | 10 ++ apps/web/e2e/agents-lifecycle.spec.ts | 38 +++- apps/web/e2e/sandbox-setup.spec.ts | 4 +- apps/web/src/App.tsx | 1 + .../src/features/dashboard/DashboardView.css | 101 +++++++++++ .../features/dashboard/DashboardView.test.tsx | 8 +- .../src/features/dashboard/DashboardView.tsx | 58 +++++-- .../dashboard/SystemObservabilityContent.tsx | 140 +++++++++++++++ .../features/dashboard/SystemRequestChart.tsx | 53 ++++++ apps/web/src/i18n/locales/en/dashboard.ts | 16 ++ apps/web/src/i18n/locales/en/pages.ts | 4 + apps/web/src/i18n/locales/zh-CN/dashboard.ts | 16 ++ apps/web/src/i18n/locales/zh-CN/pages.ts | 4 + apps/web/vite.config.ts | 1 + .../agents-api/sandbox-manager.openapi.yaml | 127 ++++++++++++++ .../agents-api/system-observability-plan.md | 94 +++++----- packages/agents-client/src/index.ts | 1 + .../agents-client/src/observability-client.ts | 42 +++++ services/agents-api/cmd/server/main.go | 18 +- .../cmd/server/observability_cleanup.go | 30 ++++ services/agents-api/internal/api/handler.go | 7 + .../internal/api/observability_read.go | 88 ++++++++++ .../internal/api/observability_read_test.go | 62 +++++++ .../internal/api/observability_requests.go | 76 ++++++++ .../internal/db/queries/runtime_history.sql | 8 +- .../agents-api/internal/db/sqlc/models.go | 85 +++++++-- .../internal/db/sqlc/runtime_history.sql.go | 14 +- .../agents-api/internal/execution/delivery.go | 2 +- .../internal/execution/dispatcher.go | 5 + .../agents-api/internal/execution/journal.go | 70 ++++++++ .../execution/observability_tools_test.go | 60 +++++++ .../internal/observability/requests.go | 134 ++++++++++++++ .../internal/observability/requests_test.go | 40 +++++ .../internal/observability/tools.go | 90 ++++++++++ .../internal/store/observability_read.go | 164 ++++++++++++++++++ .../internal/store/observability_requests.go | 63 +++++++ .../store/observability_requests_test.go | 37 ++++ .../internal/store/observability_retention.go | 29 ++++ .../internal/store/observability_tools.go | 49 ++++++ .../store/observability_tools_test.go | 55 ++++++ .../store/runtime_history_acceptance_test.go | 12 +- .../migrations/000066_observability.sql | 78 +++++++++ .../agents-api/tests/runtime_history_live.py | 2 +- services/core-console/sandbox_admin.go | 5 + services/core-console/sandbox_admin_test.go | 5 +- services/core-console/server.go | 8 +- 46 files changed, 1916 insertions(+), 98 deletions(-) create mode 100644 apps/web/src/features/dashboard/SystemObservabilityContent.tsx create mode 100644 apps/web/src/features/dashboard/SystemRequestChart.tsx create mode 100644 packages/agents-client/src/observability-client.ts create mode 100644 services/agents-api/cmd/server/observability_cleanup.go create mode 100644 services/agents-api/internal/api/observability_read.go create mode 100644 services/agents-api/internal/api/observability_read_test.go create mode 100644 services/agents-api/internal/api/observability_requests.go create mode 100644 services/agents-api/internal/execution/observability_tools_test.go create mode 100644 services/agents-api/internal/observability/requests.go create mode 100644 services/agents-api/internal/observability/requests_test.go create mode 100644 services/agents-api/internal/observability/tools.go create mode 100644 services/agents-api/internal/store/observability_read.go create mode 100644 services/agents-api/internal/store/observability_requests.go create mode 100644 services/agents-api/internal/store/observability_requests_test.go create mode 100644 services/agents-api/internal/store/observability_retention.go create mode 100644 services/agents-api/internal/store/observability_tools.go create mode 100644 services/agents-api/internal/store/observability_tools_test.go create mode 100644 services/agents-api/migrations/000066_observability.sql diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 12cb118d0..5f671aa6d 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -144,6 +144,16 @@ logs. Update this guide when architecture, ownership or generated contracts chan Comments and documentation are English. Reuse existing helpers and error mapping; split oversized components before extending them. Use `internal/obs/log` for logs. +Operator observability is a side-channel owned by Core. The authenticated +`/core/v1/observability/summary` extension reads bounded request, terminal Turn, +tool attempt, and collector aggregates; the Core Web uses it through +`packages/agents-client`. Runtime samples retain their separate public history +contract under the `observability_runtime_samples` table. Native model attempts +have a reserved table but no qualified producer; configured models, cumulative +Usage, and Turns must not be counted as model calls. Missing collector coverage +stays unavailable rather than becoming zero. See +`contracts/agents-api/system-observability-plan.md` for source and retention details. + ## Required checks Run `make check` before completion. The standalone gate includes all daemon/shared diff --git a/apps/web/e2e/agents-lifecycle.spec.ts b/apps/web/e2e/agents-lifecycle.spec.ts index 7b63cb271..e2d653a05 100644 --- a/apps/web/e2e/agents-lifecycle.spec.ts +++ b/apps/web/e2e/agents-lifecycle.spec.ts @@ -210,6 +210,36 @@ test("opens Dashboard as the default landing page", async ({ page, request }) => await expect(page.locator(".dashboard-page")).toBeVisible(); }); +test("shows operator tables and keeps incomplete request coverage unavailable", async ({ page, request }, testInfo) => { + await resetFixture(request); + await page.route("**/console/config", (route) => route.fulfill({ json: { + sandbox_admin: true, node_installer: false, node_installer_sha256: "", + } })); + const end = new Date(Math.floor(Date.now() / 60_000) * 60_000); + const start = new Date(end.getTime() - 60 * 60_000); + await page.route("**/core/v1/observability/summary?range=1h", (route) => route.fulfill({ json: { + generated_at: end.toISOString(), start: start.toISOString(), end: end.toISOString(), step_seconds: 60, + requests: [{ start: start.toISOString(), route_family: "sessions", outcome: "server_error", count: 2, + latency_sum_ms: 120, latency_bucket_counts: [0, 0, 0, 2, 0, 0, 0, 0, 0, 0, 0] }], + collector: [{ start: start.toISOString(), source: "request", attempted_count: 2, observed_count: 2, + unavailable_count: 0, timeout_count: 0, dropped_count: 0, export_failed_count: 0 }], + turns: [{ start: start.toISOString(), status: "failed", count: 1, queue_p95_ms: 50, execution_p95_ms: 400 }], + tools: [{ start: start.toISOString(), category: "command", outcome: "error", count: 1, + timed_count: 1, duration_p95_ms: 300 }], + } })); + await page.goto("/"); + await page.getByRole("tab", { name: "Observability" }).click(); + const system = page.locator(".dashboard-system"); + await expect(system.getByRole("heading", { name: "System observability" })).toBeVisible(); + await expect(system.getByRole("heading", { name: "Terminal Turns" })).toBeVisible(); + await expect(system.getByRole("heading", { name: "Actual model invocations" })).toBeVisible(); + await expect(system.getByRole("heading", { name: "Sandbox nodes" })).toBeVisible(); + await expect(system.getByText("Unavailable", { exact: true }).first()).toBeVisible(); + await expect(system.getByRole("rowheader", { name: "sessions" })).toBeVisible(); + await expect(system.getByRole("rowheader", { name: "command" })).toBeVisible(); + await attachElementScreenshot(system, testInfo, "system-observability-dashboard"); +}); + test("explains a 502 Core backend failure and opens copyable Docker recovery steps", async ({ page }) => { await page.route("**/v1/agents**", async (route) => { await route.fulfill({ @@ -2388,9 +2418,8 @@ test("presents Dashboard page-chain results and System boundaries without extra await expect(dashboard).toContainText("Latest complete paginated reads"); await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "Agents" })).toContainText("3"); await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "Sessions" })).toContainText("1"); - await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "In progress" })).toContainText("0"); - await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "Needs attention" })).toContainText("0"); - await expect(dashboard).not.toContainText("Reported aggregate tokens"); + await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "Active sandboxes" })).toBeVisible(); + await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "Reported tokens" })).toBeVisible(); await expect(dashboard).toContainText("No Sessions currently need attention."); await expect(dashboard.getByRole("button", { name: /Create agent/ })).toBeVisible(); await expect(dashboard.getByRole("button", { name: /Start session/ })).toBeVisible(); @@ -2683,6 +2712,7 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand await page.goto("/"); const dashboard = page.locator(".dashboard-page"); await expect(dashboard.getByRole("heading", { name: "Dashboard", exact: true })).toBeVisible(); + await dashboard.getByRole("tab", { name: "Observability" }).click(); const refresh = dashboard.getByRole("button", { name: "Refresh Dashboard snapshot" }); const sandboxDiagnostics = dashboard.getByRole("region", { name: "Sandbox diagnostics" }); await expect(sandboxDiagnostics).toBeVisible(); @@ -2918,6 +2948,7 @@ test("restores retained Runtime history after a Dashboard reload", async ({ page }); await page.goto("/"); + await page.getByRole("tab", { name: "Observability" }).click(); const dashboard = page.locator(".dashboard-runtime-panel"); await expect(dashboard.getByRole("group", { name: "Runtime trend source" })).toHaveCount(0); await expect(dashboard.getByLabel(/Durable · 30s; 1 Runtime targets/)).toBeVisible(); @@ -2978,6 +3009,7 @@ test("restores retained Runtime history after a Dashboard reload", async ({ page await expect(refreshedDurableCpuChart).toHaveAttribute("data-view-end", String(durableZoomEnd)); await page.reload(); + await page.getByRole("tab", { name: "Observability" }).click(); await expect(dashboard.getByRole("group", { name: "Runtime trend source" })).toHaveCount(0); await expect(dashboard.getByLabel(/Durable · 30s; 1 Runtime targets/)).toBeVisible(); await expect(dashboard.getByText("CPU usage durable trend available")).toBeAttached(); diff --git a/apps/web/e2e/sandbox-setup.spec.ts b/apps/web/e2e/sandbox-setup.spec.ts index c58cd027d..c5fd35b88 100644 --- a/apps/web/e2e/sandbox-setup.spec.ts +++ b/apps/web/e2e/sandbox-setup.spec.ts @@ -189,13 +189,15 @@ for (const operation of ["setup", "enrollment"] as const) { } test("unpaired or unavailable consoles show setup guidance without admin credentials or manager requests", async ({ page, request }) => { + await expect.poll(async () => ((await (await request.get(`${fixture}/__fixture/sandbox`)).json()) as { calls: unknown[] }).calls.length).toBeGreaterThan(0); + const initialCalls = ((await (await request.get(`${fixture}/__fixture/sandbox`)).json()) as { calls: unknown[] }).calls.length; for (const body of [{ sandbox_admin: false, node_installer: true }, null]) { await page.route("**/console/config", (route) => route.fulfill(body ? { contentType: "application/json", body: JSON.stringify(body) } : { status: 404, body: "Not found" })); await page.reload(); await page.getByRole("button", { name: "Hosted Sandbox Manager", exact: true }).click(); await expect(page.getByRole("alert")).toContainText("Sandbox administration is not configured"); await expect(page.getByLabel("Deployment admin key")).toHaveCount(0); - expect((await (await request.get(`${fixture}/__fixture/sandbox`)).json()).calls).toHaveLength(0); + expect((await (await request.get(`${fixture}/__fixture/sandbox`)).json()).calls).toHaveLength(initialCalls); } await page.route("**/console/config", (route) => route.fulfill({ contentType: "application/json", body: JSON.stringify({ sandbox_admin: true, node_installer: true, node_installer_sha256: "a".repeat(64) }) })); await page.getByRole("button", { name: "Refresh sandbox state" }).click(); diff --git a/apps/web/src/App.tsx b/apps/web/src/App.tsx index 7765467be..8464c582d 100644 --- a/apps/web/src/App.tsx +++ b/apps/web/src/App.tsx @@ -2367,6 +2367,7 @@ export function App() { runtimeCollectionError={runtimeCollectionError} runtimeCollectionHasSnapshot={runtimeCollectionHasSnapshot} loadRuntimeHistory={loadDashboardRuntimeHistory} + coreBaseUrl={connection.baseUrl} onRefresh={refreshDashboard} onCreateAgent={openAgentSetup} onStartSession={() => openSessionSetup()} diff --git a/apps/web/src/features/dashboard/DashboardView.css b/apps/web/src/features/dashboard/DashboardView.css index 5364e670c..92ff876f4 100644 --- a/apps/web/src/features/dashboard/DashboardView.css +++ b/apps/web/src/features/dashboard/DashboardView.css @@ -5,6 +5,79 @@ overflow: hidden; } +.dashboard-system { + display: grid; + gap: 16px; + min-width: 0; + margin-top: 18px; +} + +.dashboard-system-header, +.dashboard-system-controls, +.dashboard-system-metrics { + display: flex; + align-items: center; +} + +.dashboard-system-header { + justify-content: space-between; + gap: 12px; +} + +.dashboard-system-header h2, +.dashboard-system-tables h3 { margin: 0; } +.dashboard-system-header p, +.dashboard-system-source, +.dashboard-system-tables p { margin: 4px 0 0; color: var(--fg-muted); font-size: 12px; } +.dashboard-system-controls { gap: 4px; } +.dashboard-system-controls button { + padding: 5px 9px; + border: 1px solid var(--line); + border-radius: 6px; + background: var(--surface); + color: var(--fg-muted); + cursor: pointer; +} +.dashboard-system-controls button[aria-pressed="true"] { color: var(--fg); border-color: var(--fg-muted); } +.dashboard-system-metrics { align-items: stretch; flex-wrap: wrap; gap: 8px; } +.dashboard-system-metrics > div { + display: grid; + flex: 1 1 150px; + gap: 5px; + min-width: 0; + padding: 13px; + border: 1px solid var(--line); + border-radius: 8px; + background: var(--surface); +} +.dashboard-system-metrics small, +.dashboard-system-metrics span { color: var(--fg-muted); font-size: 11px; } +.dashboard-system-metrics strong { font-size: 20px; font-variant-numeric: tabular-nums; } +.dashboard-system-tables { display: grid; grid-template-columns: repeat(2, minmax(0, 1fr)); gap: 12px; } +.dashboard-system-request-chart { + min-width: 0; + margin: 0; + padding: 14px; + border: 1px solid var(--line); + border-radius: 8px; + background: var(--surface); +} +.dashboard-system-request-chart figcaption { display: flex; align-items: baseline; justify-content: space-between; gap: 12px; font-size: 13px; } +.dashboard-system-request-chart figcaption span { color: var(--fg-muted); font-size: 11px; } +.dashboard-system-request-chart svg { display: block; width: 100%; height: 150px; margin-top: 9px; color: var(--fg); } +.dashboard-system-tables > section { min-width: 0; padding: 14px; border: 1px solid var(--line); border-radius: 8px; background: var(--surface); } +.dashboard-system-nodes { grid-column: 1 / -1; } +.dashboard-system-table-scroll { overflow-x: auto; margin-top: 10px; } +.dashboard-system-table-scroll table { width: 100%; border-collapse: collapse; font-size: 12px; } +.dashboard-system-table-scroll th, +.dashboard-system-table-scroll td { padding: 8px 7px; border-bottom: 1px solid var(--line); text-align: right; white-space: nowrap; } +.dashboard-system-table-scroll th:first-child { text-align: left; } +.dashboard-system-table-scroll thead th { color: var(--fg-muted); font-weight: 500; } +@media (max-width: 760px) { + .dashboard-system-header { align-items: flex-start; flex-direction: column; } + .dashboard-system-tables { grid-template-columns: 1fr; } +} + .dashboard-header > div:first-child { min-width: 0; } @@ -27,6 +100,34 @@ overflow-y: auto; } +.dashboard-view-tabs { + display: flex; + width: fit-content; + margin-bottom: 14px; + padding: 3px; + border: 1px solid var(--line); + border-radius: 8px; + background: var(--surface); + gap: 3px; +} + +.dashboard-view-tabs button { + padding: 7px 13px; + border: 0; + border-radius: 5px; + background: transparent; + color: var(--fg-muted); + cursor: pointer; + font-size: 12px; + font-weight: 600; +} + +.dashboard-view-tabs button[aria-selected="true"] { + background: var(--surface-raised, var(--surface)); + color: var(--fg); + box-shadow: var(--shadow-control); +} + .dashboard-overview, .dashboard-panel { min-width: 0; diff --git a/apps/web/src/features/dashboard/DashboardView.test.tsx b/apps/web/src/features/dashboard/DashboardView.test.tsx index 4ad132758..7acd92461 100644 --- a/apps/web/src/features/dashboard/DashboardView.test.tsx +++ b/apps/web/src/features/dashboard/DashboardView.test.tsx @@ -128,13 +128,14 @@ describe("Dashboard loaded-result presentation", () => { it("keeps a Runtime-only 503 scoped to the optional observation feature", () => { const html = render({ + initialTab: "observability", runtimeSnapshot: null, runtimeCollectionState: "failed", runtimeCollectionError: "Agent core request failed (503).", runtimeCollectionHasSnapshot: false, }); - expect(html).toContain("Runtime: Agent core request failed (503)."); + expect(html).toContain("Agent core request failed (503)."); expect(html).toContain("Runtime observations unavailable"); expect(html).not.toContain("Agent Core backend is not ready"); expect(html).not.toContain("Core backend is offline"); @@ -239,6 +240,7 @@ describe("Dashboard loaded-result presentation", () => { memory: { usage_bytes: 536_870_912, limit_bytes: 2_147_483_648 }, }; const html = render({ + initialTab: "observability", runtimeSnapshot: { sessions: [hosted], observations: [observation], loadedAt: 1_700_000_100_000 }, }); @@ -250,6 +252,7 @@ describe("Dashboard loaded-result presentation", () => { expect(html).toContain("Sandbox diagnostics"); expect(html).toContain("Memory ≥80%"); const withoutMemory = render({ + initialTab: "observability", runtimeSnapshot: { sessions: [hosted], observations: [{ ...observation, memory: { usage_bytes: null, limit_bytes: null } }], loadedAt: 1_700_000_100_000 }, }); expect(withoutMemory).toMatch(/Memory ≥80%<\/small>Unavailable<\/strong>/); @@ -361,13 +364,14 @@ describe("Dashboard loaded-result presentation", () => { memory: null, }; const html = render({ + initialTab: "observability", runtimeSnapshot: { sessions: [runtimeSession], observations: [observation], loadedAt: 1_700_000_100_000 }, runtimeCollectionState: "failed", runtimeCollectionError: "Runtime refresh failed", runtimeCollectionHasSnapshot: true, }); - expect(html).toContain("Runtime: Runtime refresh failed"); + expect(html).toContain("Runtime refresh failed"); expect(html).toContain("Live · loading history"); expect(html).not.toContain("Live · 30s"); }); diff --git a/apps/web/src/features/dashboard/DashboardView.tsx b/apps/web/src/features/dashboard/DashboardView.tsx index 5ac9a7cd2..5a97c4505 100644 --- a/apps/web/src/features/dashboard/DashboardView.tsx +++ b/apps/web/src/features/dashboard/DashboardView.tsx @@ -6,21 +6,26 @@ import { RefreshCw, Rows3, } from "lucide-react"; -import { useMemo, type ReactNode } from "react"; +import { useEffect, useMemo, useState, type ReactNode } from "react"; import { useTranslation } from "react-i18next"; -import type { AgentSession, SavedAgent } from "@agents-core-web/agents-client"; +import { SandboxAdminClient, type AgentSession, type SandboxNode, type SavedAgent } from "@agents-core-web/agents-client"; import { StatusIcon, type StatusKind } from "../../components/StatusIcon"; import { backendFailureStatus } from "../../lib/core-readiness"; +import { isLocalProxyBaseUrl } from "../../lib/connection"; +import { sandboxConsoleConfig } from "../sandbox/console-config"; import { buildDashboardSnapshot, buildRuntimeDashboardModel, + formatDashboardBytes, formatDashboardTimestamp, + formatDashboardTokens, type DashboardCollectionState, type DashboardSessionRow, } from "./dashboard-model"; import { RuntimeObservabilityContent, type RuntimeHistoryLoader } from "./RuntimeObservabilityContent"; +import { SystemObservabilityContent } from "./SystemObservabilityContent"; import type { RuntimeDashboardSnapshot } from "./runtime-snapshot"; import "./DashboardView.css"; @@ -38,6 +43,8 @@ export interface DashboardViewProps { runtimeCollectionError: string | null; runtimeCollectionHasSnapshot: boolean; loadRuntimeHistory: RuntimeHistoryLoader; + coreBaseUrl?: string; + initialTab?: "overview" | "observability"; onRefresh: () => void; onCreateAgent: () => void; onStartSession: () => void; @@ -247,6 +254,8 @@ export function DashboardView({ runtimeCollectionError, runtimeCollectionHasSnapshot, loadRuntimeHistory, + coreBaseUrl = "/v1", + initialTab = "overview", onRefresh, onCreateAgent, onStartSession, @@ -257,6 +266,25 @@ export function DashboardView({ }: DashboardViewProps) { const { t, i18n } = useTranslation("pages"); const locale = i18n.resolvedLanguage; + const [tab, setTab] = useState<"overview" | "observability">(initialTab); + const [nodes, setNodes] = useState(null); + const [nodeRefresh, setNodeRefresh] = useState(0); + useEffect(() => { + const timer = window.setInterval(() => setNodeRefresh((value) => value + 1), 30_000); + return () => window.clearInterval(timer); + }, []); + useEffect(() => { + if (!isLocalProxyBaseUrl(coreBaseUrl)) { setNodes(null); return; } + const controller = new AbortController(); + void (async () => { + const config = await sandboxConsoleConfig(controller.signal); + if (!config?.sandbox_admin) { if (!controller.signal.aborted) setNodes(null); return; } + const client = new SandboxAdminClient({ baseUrl: "/core/v1/sandbox" }); + const result = await client.listNodes({ signal: controller.signal }); + if (!controller.signal.aborted) setNodes(result.data); + })().catch(() => { if (!controller.signal.aborted) setNodes(null); }); + return () => controller.abort(); + }, [coreBaseUrl, nodeRefresh]); const snapshot = useMemo(() => buildDashboardSnapshot(agents, sessions, 6, 5), [agents, sessions]); const runtimeModel = useMemo(() => runtimeSnapshot ? buildRuntimeDashboardModel(runtimeSnapshot.sessions, runtimeSnapshot.observations) @@ -302,7 +330,7 @@ export function DashboardView({ + + + {tab === "overview" ? <>
@@ -371,16 +404,16 @@ export function DashboardView({
- - 0} - /> + + + + + node.online && node.provider_ready).length}/${nodes.length}`} detail={t("dashboard.nodeSnapshot")} />
+ : null} + {tab === "observability" ? <>
@@ -393,6 +426,7 @@ export function DashboardView({ ) : null}
+ {runtimeCollectionState === "failed" && runtimeCollectionError ?

{runtimeCollectionError}

: null} {!runtimeAvailable || !runtimeSnapshot || !runtimeModel ? (

+ + : null} + {tab === "overview" ? <>
@@ -476,6 +513,7 @@ export function DashboardView({

)}
+ : null}
); diff --git a/apps/web/src/features/dashboard/SystemObservabilityContent.tsx b/apps/web/src/features/dashboard/SystemObservabilityContent.tsx new file mode 100644 index 000000000..cd5391b88 --- /dev/null +++ b/apps/web/src/features/dashboard/SystemObservabilityContent.tsx @@ -0,0 +1,140 @@ +import { useEffect, useMemo, useState } from "react"; +import { useTranslation } from "react-i18next"; +import { + ObservabilityAdminClient, + type OperatorMetricsRange, + type OperatorMetricsSummary, + type RequestMetricBucket, + type SandboxNode, +} from "@agents-core-web/agents-client"; + +import { isLocalProxyBaseUrl } from "../../lib/connection"; +import { sandboxConsoleConfig } from "../sandbox/console-config"; +import { formatDashboardBytes } from "./dashboard-model"; +import { SystemRequestChart } from "./SystemRequestChart"; + +const latencyBounds = [10, 25, 50, 100, 250, 500, 1000, 2500, 5000, 10000] as const; + +function percentile95(rows: readonly RequestMetricBucket[]): number | null { + const counts = Array.from({ length: 11 }, (_, index) => rows.reduce((sum, row) => sum + (row.latency_bucket_counts[index] ?? 0), 0)); + const total = counts.reduce((sum, count) => sum + count, 0); + if (total === 0) return null; + const target = Math.ceil(total * .95); + let current = 0; + for (let index = 0; index < counts.length; index += 1) { + current += counts[index]!; + if (current >= target) return latencyBounds[index] ?? 10000; + } + return null; +} + +interface RouteSummary { + family: string; + count: number; + errors: number; + latency: number | null; +} + +function routeSummaries(rows: readonly RequestMetricBucket[]): RouteSummary[] { + const families = new Map(); + for (const row of rows) { + const previous = families.get(row.route_family) ?? []; + previous.push(row); + families.set(row.route_family, previous); + } + return [...families].map(([family, group]) => ({ + family, + count: group.reduce((sum, row) => sum + row.count, 0), + errors: group.filter((row) => row.outcome === "server_error").reduce((sum, row) => sum + row.count, 0), + latency: percentile95(group), + })).sort((left, right) => right.count - left.count); +} + +export function SystemObservabilityContent({ coreBaseUrl, nodes }: { coreBaseUrl: string; nodes: SandboxNode[] | null }) { + const { t, i18n } = useTranslation("dashboard"); + const locale = i18n.resolvedLanguage; + const [range, setRange] = useState("1h"); + const [refresh, setRefresh] = useState(0); + const [summary, setSummary] = useState(null); + const [state, setState] = useState<"loading" | "ready" | "unavailable" | "failed">("loading"); + useEffect(() => { + const timer = window.setInterval(() => setRefresh((value) => value + 1), 30_000); + return () => window.clearInterval(timer); + }, []); + useEffect(() => { + const controller = new AbortController(); + if (!isLocalProxyBaseUrl(coreBaseUrl)) { + setSummary(null); + setState("unavailable"); + return () => controller.abort(); + } + setState("loading"); + setSummary(null); + void (async () => { + const config = await sandboxConsoleConfig(controller.signal); + if (!config?.sandbox_admin) { + if (!controller.signal.aborted) setState("unavailable"); + return; + } + const metricsClient = new ObservabilityAdminClient({ baseUrl: "/core/v1/observability" }); + const metrics = await metricsClient.retrieveSummary(range, { signal: controller.signal }); + if (controller.signal.aborted) return; + setSummary(metrics); + setState("ready"); + })().catch(() => { if (!controller.signal.aborted) setState("failed"); }); + return () => controller.abort(); + }, [coreBaseUrl, range, refresh]); + + const routes = useMemo(() => routeSummaries(summary?.requests ?? []), [summary]); + const total = routes.reduce((sum, row) => sum + row.count, 0); + const serverErrors = routes.reduce((sum, row) => sum + row.errors, 0); + const p95 = percentile95(summary?.requests ?? []); + const durationMinutes = range === "1h" ? 60 : range === "6h" ? 360 : 1440; + const expectedBuckets = durationMinutes / ((summary?.step_seconds ?? 60) / 60); + const heartbeatBuckets = new Set((summary?.collector ?? []).filter((row) => row.source === "request").map((row) => row.start)).size; + const completeRequestCoverage = summary !== null && heartbeatBuckets >= expectedBuckets; + const requestRate = completeRequestCoverage ? total / durationMinutes : null; + const errorRate = completeRequestCoverage && total > 0 ? serverErrors / total * 100 : null; + const collector = summary?.collector ?? []; + const dropped = collector.reduce((sum, row) => sum + row.dropped_count + row.export_failed_count, 0); + const onlineNodes = nodes?.filter((node) => node.online && node.provider_ready).length; + const nodeCapacity = nodes?.reduce((sum, node) => sum + node.max_active, 0); + const number = (value: number) => value.toLocaleString(locale); + const unavailable = t("system.unavailableValue"); + + return
+
+

{t("system.title")}

{t("system.subtitle")}

+
+ {(["1h", "6h", "24h"] as const).map((option) => )} +
+
+

{state === "loading" ? t("system.loading") : state === "failed" ? t("system.stale") : state === "unavailable" ? t("system.adminUnavailable") : t("system.source")}

+
+
{t("system.requestRate")}{requestRate === null ? unavailable : `${number(Math.round(requestRate))}/min`}{t("system.requestCount", { count: total })}
+
{t("system.serverErrorRate")}{errorRate === null ? unavailable : `${errorRate.toFixed(1)}%`}{t("system.serverErrorCount", { count: serverErrors })}
+
{t("system.latencyP95")}{!completeRequestCoverage || p95 === null ? unavailable : p95 === 10000 ? ">10 s" : `${number(p95)} ms`}{t("system.histogramEstimate")}
+
{t("system.collectorCoverage")}{summary === null ? unavailable : `${number(heartbeatBuckets)}/${number(expectedBuckets)}`}{t("system.collectorLoss", { count: dropped })}
+
{t("system.nodes")}{nodes === null ? unavailable : `${number(onlineNodes ?? 0)}/${number(nodes.length)}`}{nodeCapacity === undefined ? unavailable : t("system.nodeCapacity", { count: nodeCapacity })}
+
+ +
+

{t("system.routeTable")}

{t("system.coveredBuckets", { covered: heartbeatBuckets, total: expectedBuckets })}

+ {routes.length ?
{routes.map((row) => )}
{t("system.route")}{t("system.requests")}{t("system.errors")}{t("system.p95")}
{row.family}{number(row.count)}{number(row.errors)}{row.latency === null ? unavailable : row.latency === 10000 ? ">10 s" : `${number(row.latency)} ms`}
:

{t("system.noRequestSamples")}

} +
+

{t("system.collectorTable")}

+ {collector.length ?
{[...new Set(collector.map((row) => row.source))].map((source) => { const rows = collector.filter((row) => row.source === source); return ; })}
{t("system.sourceName")}{t("system.attempted")}{t("system.observed")}{t("system.lost")}
{source}{number(rows.reduce((sum, row) => sum + row.attempted_count, 0))}{number(rows.reduce((sum, row) => sum + row.observed_count, 0))}{number(rows.reduce((sum, row) => sum + row.dropped_count + row.export_failed_count, 0))}
:

{t("system.noCollectorSamples")}

} +
+

{t("system.turnTable")}

+ {summary?.turns.length ?
{[...new Set(summary.turns.map((row) => row.status))].map((status) => { const rows = summary.turns.filter((row) => row.status === status); const count = rows.reduce((sum, row) => sum + row.count, 0); const queue = rows.reduce((values, row) => row.queue_p95_ms === null ? values : [...values, row.queue_p95_ms], []); const execution = rows.reduce((values, row) => row.execution_p95_ms === null ? values : [...values, row.execution_p95_ms], []); return ; })}
{t("system.turnStatus")}{t("system.turnCount")}{t("system.queueP95")}{t("system.executionP95")}
{status}{number(count)}{queue.length ? `${number(Math.round(Math.max(...queue)))} ms` : unavailable}{execution.length ? `${number(Math.round(Math.max(...execution)))} ms` : unavailable}
:

{t("system.noTurns")}

} +
+

{t("system.toolTable")}

+ {summary?.tools.length ?
{summary.tools.map((row, index) => )}
{t("system.toolCategory")}{t("system.toolOutcome")}{t("system.toolCount")}{t("system.toolTiming")}
{row.category}{row.outcome}{number(row.count)}{row.duration_p95_ms === null ? unavailable : `${number(Math.round(row.duration_p95_ms))} ms · ${number(row.timed_count)}/${number(row.count)}`}
:

{t("system.noToolAttempts")}

} +
+

{t("system.modelTable")}

{t("system.modelUnavailable")}

+

{t("system.nodeTable")}

+ {nodes?.length ?
{nodes.map((node) => )}
{t("system.node")}{t("system.status")}{t("system.active")}{t("system.retained")}{t("system.cores")}{t("system.availableMemory")}
{node.name}{t(node.online && node.provider_ready ? "system.online" : "system.offline")}{number(node.active)} / {number(node.max_active)}{number(node.retained)}{node.cpu_count === null ? unavailable : number(node.cpu_count)}{node.available_memory_bytes === null ? unavailable : formatDashboardBytes(node.available_memory_bytes)}
:

{nodes === null ? t("system.nodeUnavailable") : t("system.noNodes")}

} +
+
+
; +} diff --git a/apps/web/src/features/dashboard/SystemRequestChart.tsx b/apps/web/src/features/dashboard/SystemRequestChart.tsx new file mode 100644 index 000000000..1822ff44e --- /dev/null +++ b/apps/web/src/features/dashboard/SystemRequestChart.tsx @@ -0,0 +1,53 @@ +import { useMemo } from "react"; +import { useTranslation } from "react-i18next"; +import type { OperatorMetricsSummary } from "@agents-core-web/agents-client"; + +export function SystemRequestChart({ summary }: { summary: OperatorMetricsSummary | null }) { + const { t, i18n } = useTranslation("dashboard"); + const points = useMemo(() => { + if (!summary) return []; + const start = Date.parse(summary.start); + const end = Date.parse(summary.end); + const stepMS = summary.step_seconds * 1_000; + const requests = new Map(); + for (const row of summary.requests) { + const at = Date.parse(row.start); + const prior = requests.get(at) ?? { total: 0, errors: 0 }; + prior.total += row.count; + if (row.outcome === "server_error") prior.errors += row.count; + requests.set(at, prior); + } + const covered = new Set(summary.collector.filter((row) => row.source === "request").map((row) => Date.parse(row.start))); + const result = []; + for (let at = start; at < end; at += stepMS) { + result.push({ at, value: covered.has(at) ? requests.get(at) ?? { total: 0, errors: 0 } : null }); + } + return result; + }, [summary]); + const known = points.filter((point) => point.value !== null); + const maximum = Math.max(1, ...known.map((point) => point.value?.total ?? 0)); + const width = 780; + const height = 150; + const left = 28; + const baseline = 124; + const plotWidth = width - left - 8; + const slot = plotWidth / Math.max(1, points.length); + return
+
{t("system.requestTrend")}{t("system.requestTrendDetail", { covered: known.length, total: points.length })}
+ {known.length ? + + {points.map((point, index) => { + if (point.value === null) return null; + const x = left + index * slot + slot * .12; + const barWidth = Math.max(1, slot * .75); + const barHeight = Math.max(1, point.value.total / maximum * 100); + const errorHeight = Math.max(0, point.value.errors / maximum * 100); + return + {`${new Date(point.at).toLocaleString(i18n.resolvedLanguage)}: ${point.value.total} ${t("system.requests")}, ${point.value.errors} ${t("system.errors")}`} + + {errorHeight > 0 ? : null} + ; + })} + :

{t("system.noRequestSamples")}

} +
; +} diff --git a/apps/web/src/i18n/locales/en/dashboard.ts b/apps/web/src/i18n/locales/en/dashboard.ts index 4415511ff..dcf3d0e15 100644 --- a/apps/web/src/i18n/locales/en/dashboard.ts +++ b/apps/web/src/i18n/locales/en/dashboard.ts @@ -1,4 +1,20 @@ export const dashboard = { + system: { + title: "System observability", subtitle: "Completed HTTP requests, collection coverage, and sandbox nodes", + range: "System metrics range", loading: "Loading operator metrics…", stale: "Operator metrics unavailable · retrying", + adminUnavailable: "Deployment administrator metrics are unavailable on this connection.", source: "Core operator metrics · completed time buckets", + unavailableValue: "Unavailable", requestRate: "Request rate", requestCount: "{{count}} observed completed requests", + serverErrorRate: "HTTP 5xx rate", serverErrorCount: "{{count}} observed server errors", latencyP95: "Latency p95", histogramEstimate: "Estimated from fixed latency buckets", + collectorCoverage: "Request collector buckets", collectorLoss: "{{count}} dropped or failed writes", nodes: "Ready nodes", nodeCapacity: "{{count}} active slots configured", + routeTable: "Request routes", coveredBuckets: "Observed buckets: {{covered}}/{{total}}", route: "Route family", requests: "Requests", errors: "5xx", p95: "p95", + requestTrend: "Request volume", requestTrendDetail: "{{covered}}/{{total}} buckets covered · gaps are unavailable", requestTrendLabel: "Request volume over time; {{covered}} of {{total}} buckets covered", + collectorTable: "Collector health", sourceName: "Source", attempted: "Attempted", observed: "Observed", lost: "Dropped / failed", + turnTable: "Terminal Turns", turnStatus: "Outcome", turnCount: "Turns", queueP95: "Worst bucket queue p95", executionP95: "Worst bucket run p95", noTurns: "No terminal Turns in this range.", + toolTable: "Actual tool attempts", toolCategory: "Category", toolOutcome: "Outcome", toolCount: "Attempts", toolTiming: "Bucket p95 · timed coverage", noToolAttempts: "No terminal tool attempt observations in this range.", + modelTable: "Actual model invocations", modelUnavailable: "Invocation-level collection is not yet available from every harness. Configured models and Turn usage are not counted as calls.", + nodeTable: "Sandbox nodes", node: "Node", status: "Status", active: "Active / capacity", retained: "Retained", cores: "CPU cores", availableMemory: "Available memory", online: "Ready", offline: "Unavailable", + noRequestSamples: "No request buckets collected in this range.", noCollectorSamples: "No collector coverage recorded in this range.", noNodes: "No sandbox nodes registered.", nodeUnavailable: "Node data unavailable.", + }, sandbox: { title: "Sandbox diagnostics", subtitle: "Current allocation identity, sampling coverage, and memory pressure", current: "Current complete snapshot", retained: "Last complete snapshot · refresh failed", diff --git a/apps/web/src/i18n/locales/en/pages.ts b/apps/web/src/i18n/locales/en/pages.ts index 9c6de3a3d..be3773f72 100644 --- a/apps/web/src/i18n/locales/en/pages.ts +++ b/apps/web/src/i18n/locales/en/pages.ts @@ -1,5 +1,9 @@ export const pages = { dashboard: { + viewTabs: "Dashboard views", overviewTab: "Overview", observabilityTab: "Observability", + activeSandboxes: "Active sandboxes", currentRuntimeSnapshot: "Current Runtime snapshot", totalTokens: "Reported tokens", reportedUsage: "Session usage with coverage", + cpuCapacity: "Sandbox CPU capacity", measuredCores: "Cores with a current observation", memoryLimit: "Sandbox memory limit", measuredLimit: "Limits with a current observation", + readyNodes: "Ready nodes", nodeSnapshot: "Deployment node heartbeat", title: "Dashboard", subtitle: "Agents and Sessions that may need your attention.", refresh: "Refresh", refreshing: "Refreshing…", refreshLabel: "Refresh Dashboard snapshot", snapshotIncomplete: "Snapshot incomplete", snapshotStale: "Using the last successful snapshot", snapshotRefreshing: "Refreshing snapshot", snapshotReady: "Snapshot ready", snapshotDetail: "Latest complete paginated reads · not a live Core total or runtime-readiness signal", dataSources: "Dashboard data sources", agents: "Agents", sessions: "Sessions", runtime: "Runtime", unavailable: "Unavailable", savedDefinitions: "Saved definitions", inSnapshot: "In this snapshot", inProgress: "In progress", reportedStatus: "Core-reported status", needsAttention: "Needs attention", attentionDetail: "Requires action or failed", diff --git a/apps/web/src/i18n/locales/zh-CN/dashboard.ts b/apps/web/src/i18n/locales/zh-CN/dashboard.ts index 7c9ca0cf5..010f4a60d 100644 --- a/apps/web/src/i18n/locales/zh-CN/dashboard.ts +++ b/apps/web/src/i18n/locales/zh-CN/dashboard.ts @@ -1,4 +1,20 @@ export const dashboard = { + system: { + title: "系统观测", subtitle: "已完成的 HTTP 请求、采集覆盖与 Sandbox 节点", + range: "系统指标时间范围", loading: "正在加载管理员指标…", stale: "管理员指标不可用 · 正在重试", + adminUnavailable: "当前连接无法读取部署管理员指标。", source: "Core 管理员指标 · 已完成的时间桶", + unavailableValue: "不可用", requestRate: "请求速率", requestCount: "已观测 {{count}} 个完成请求", + serverErrorRate: "HTTP 5xx 比例", serverErrorCount: "已观测 {{count}} 个服务端错误", latencyP95: "延迟 p95", histogramEstimate: "按固定延迟桶估算", + collectorCoverage: "请求采集时间桶", collectorLoss: "{{count}} 次丢弃或写入失败", nodes: "就绪节点", nodeCapacity: "配置 {{count}} 个活动名额", + routeTable: "请求分类", coveredBuckets: "已观测时间桶:{{covered}}/{{total}}", route: "路由分类", requests: "请求数", errors: "5xx", p95: "p95", + requestTrend: "请求量", requestTrendDetail: "覆盖 {{covered}}/{{total}} 个时间桶 · 缺口表示不可用", requestTrendLabel: "请求量趋势;覆盖 {{covered}}/{{total}} 个时间桶", + collectorTable: "采集健康", sourceName: "来源", attempted: "尝试", observed: "成功", lost: "丢弃/失败", + turnTable: "终态 Turn", turnStatus: "结果", turnCount: "Turn 数", queueP95: "最差时间桶排队 p95", executionP95: "最差时间桶执行 p95", noTurns: "该时段没有终态 Turn。", + toolTable: "实际工具调用", toolCategory: "类别", toolOutcome: "结果", toolCount: "调用数", toolTiming: "时间桶 p95 · 时长覆盖", noToolAttempts: "该时段没有终态工具调用观测。", + modelTable: "实际模型调用", modelUnavailable: "各 Harness 尚未提供完整的单次调用采集。不会将配置模型或 Turn 用量当作调用次数。", + nodeTable: "Sandbox 节点", node: "节点", status: "状态", active: "活动/容量", retained: "保留", cores: "CPU 核数", availableMemory: "可用内存", online: "就绪", offline: "不可用", + noRequestSamples: "该时段尚无请求时间桶。", noCollectorSamples: "该时段尚无采集覆盖记录。", noNodes: "尚未注册 Sandbox 节点。", nodeUnavailable: "节点数据不可用。", + }, sandbox: { title: "Sandbox 诊断", subtitle: "按分配身份查看采样覆盖、生命周期与内存压力", current: "当前完整快照", retained: "上次完整快照 · 刷新失败", diff --git a/apps/web/src/i18n/locales/zh-CN/pages.ts b/apps/web/src/i18n/locales/zh-CN/pages.ts index 0a1bdc8c5..9a703e8eb 100644 --- a/apps/web/src/i18n/locales/zh-CN/pages.ts +++ b/apps/web/src/i18n/locales/zh-CN/pages.ts @@ -1,5 +1,9 @@ export const pages = { dashboard: { + viewTabs: "大盘视图", overviewTab: "概览", observabilityTab: "观测大盘", + activeSandboxes: "活跃 Sandbox", currentRuntimeSnapshot: "当前 Runtime 快照", totalTokens: "已报告 Token", reportedUsage: "有覆盖的会话用量", + cpuCapacity: "Sandbox CPU 容量", measuredCores: "有当前观测的核数", memoryLimit: "Sandbox 内存上限", measuredLimit: "有当前观测的上限", + readyNodes: "就绪节点", nodeSnapshot: "部署节点心跳", title: "概览", subtitle: "查看可能需要关注的智能体和会话。", refresh: "刷新", refreshing: "刷新中…", refreshLabel: "刷新概览快照", snapshotIncomplete: "快照不完整", snapshotStale: "正在使用最近一次成功快照", snapshotRefreshing: "正在刷新快照", snapshotReady: "快照已就绪", snapshotDetail: "最近一次完整分页读取 · 不代表 Core 实时总量或运行时就绪", dataSources: "概览数据源", agents: "智能体", sessions: "会话", runtime: "运行时", unavailable: "不可用", savedDefinitions: "已保存定义", inSnapshot: "当前快照", inProgress: "进行中", reportedStatus: "Core 报告状态", needsAttention: "需要关注", attentionDetail: "需要操作或已失败", diff --git a/apps/web/vite.config.ts b/apps/web/vite.config.ts index 6ec76098e..b13b470d6 100644 --- a/apps/web/vite.config.ts +++ b/apps/web/vite.config.ts @@ -46,6 +46,7 @@ export default defineConfig(({ command, mode }) => { server: { proxy: { "/core/v1/sandbox": { target, changeOrigin: true }, + "/core/v1/observability": { target, changeOrigin: true }, "/v1": { target, changeOrigin: true, diff --git a/contracts/agents-api/sandbox-manager.openapi.yaml b/contracts/agents-api/sandbox-manager.openapi.yaml index e9d7d2807..d74425b6a 100644 --- a/contracts/agents-api/sandbox-manager.openapi.yaml +++ b/contracts/agents-api/sandbox-manager.openapi.yaml @@ -7,6 +7,33 @@ definitions: rotate: type: boolean type: object + api.OperatorMetricsResponse: + properties: + collector: + items: + $ref: '#/definitions/store.CollectorMetricBucket' + type: array + end: + type: string + generated_at: + type: string + requests: + items: + $ref: '#/definitions/store.RequestMetricBucket' + type: array + start: + type: string + step_seconds: + type: integer + tools: + items: + $ref: '#/definitions/store.ToolMetricBucket' + type: array + turns: + items: + $ref: '#/definitions/store.TurnMetricBucket' + type: array + type: object api.ProjectAPIKeyList: properties: data: @@ -102,6 +129,25 @@ definitions: revoked_at: type: string type: object + store.CollectorMetricBucket: + properties: + attempted_count: + type: integer + dropped_count: + type: integer + export_failed_count: + type: integer + observed_count: + type: integer + source: + type: string + start: + type: string + timeout_count: + type: integer + unavailable_count: + type: integer + type: object store.IssuedExecutorCredential: properties: environment_id: @@ -139,6 +185,23 @@ definitions: revoked_at: type: string type: object + store.RequestMetricBucket: + properties: + count: + type: integer + latency_bucket_counts: + items: + type: integer + type: array + latency_sum_ms: + type: integer + outcome: + type: string + route_family: + type: string + start: + type: string + type: object store.ResourceOwner: properties: api_key: @@ -300,6 +363,34 @@ definitions: maintenance: type: boolean type: object + store.ToolMetricBucket: + properties: + category: + type: string + count: + type: integer + duration_p95_ms: + type: number + outcome: + type: string + start: + type: string + timed_count: + type: integer + type: object + store.TurnMetricBucket: + properties: + count: + type: integer + execution_p95_ms: + type: number + queue_p95_ms: + type: number + start: + type: string + status: + type: string + type: object store.WriteOperation: properties: action: @@ -451,6 +542,42 @@ paths: summary: Revoke an Environment executor credential tags: - Environment Executor + /core/v1/observability/summary: + get: + description: Deployment administrator only. Returns fixed-range aggregate requests and collector coverage without paths, payloads or project identifiers. + parameters: + - description: 1h, 6h, or 24h (default 1h) + in: query + name: range + type: string + produces: + - application/json + responses: + "200": + description: OK + schema: + $ref: '#/definitions/api.OperatorMetricsResponse' + "400": + description: Bad Request + schema: + $ref: '#/definitions/v1.ErrorResponse' + "401": + description: Unauthorized + schema: + $ref: '#/definitions/v1.ErrorResponse' + "500": + description: Internal Server Error + schema: + $ref: '#/definitions/v1.ErrorResponse' + "503": + description: Service Unavailable + schema: + $ref: '#/definitions/v1.ErrorResponse' + security: + - DeploymentAdminAuth: [] + summary: Retrieve bounded operator observability metrics + tags: + - Observability /core/v1/project-api-keys/{binding_digest}: get: description: Core extension requiring deployment administrator authority. The binding digest selects only a statically configured caller; it does not authenticate. Returns safe metadata, never key secrets or token digests. Derived keys freeze that caller's principal and stop authenticating when the static parent is removed or rebound. diff --git a/contracts/agents-api/system-observability-plan.md b/contracts/agents-api/system-observability-plan.md index 1f59827c4..0c52c0b1f 100644 --- a/contracts/agents-api/system-observability-plan.md +++ b/contracts/agents-api/system-observability-plan.md @@ -1,8 +1,9 @@ # System observability collection plan -Status: collection inventory and next-phase design. The current Dashboard release -uses existing Runtime observations and Session Usage only. Request, model, and tool -panels must not be shown as measured until the instruments below are implemented. +Status: implementation inventory for the operator Dashboard. Runtime history, +request buckets, terminal Turns, tool attempts, and node state have independent +sources. The model-attempt table is reserved but has no writer; its panel must +remain unavailable until native per-attempt events are qualified. ## Existing evidence @@ -12,49 +13,64 @@ panels must not be shown as measured until the instruments below are implemented | CPU time and capacity | Provider observation | Current and periodic history when sampling is available | Cumulative time; utilization needs two samples from one compute incarnation | | Memory usage and limit | Provider observation | Current and periodic history when reported | Point-in-time guest/container memory, not host memory | | Token input/output | Measured Session Usage | Public current Session and periodic Runtime history | Reported canonical usage; absent usage remains unknown | -| Request count, latency, errors | No common metric | Not available | Do not derive from Session count or HTTP page loads | -| Model distribution and latency | Saved Agent model is available, terminal call records are not aggregated | Configuration only | Do not label configured models as invoked models | -| Tool calls and failures | Durable Turn events exist, no bounded aggregate read | Partial event evidence | Do not count tools by Agent declarations | -| Host, database, queue health | No tenant-safe operator projection | Not available | Keep out of Session-scoped charts | +| Request count, latency, errors | Completed API responses, stored in minute buckets | Fixed route families, methods, outcomes, and latency bins; collector heartbeat tracks quiet intervals | Transport health, not Turn success | +| Terminal Turn outcome and duration | Persisted `turns` terminal timestamps | Completed, failed, and cancelled Turns | Execution outcome, not HTTP success | +| Model distribution and latency | Native per-attempt evidence is not normalized across harnesses | Unavailable | Do not label configured models, cumulative Usage, or Turns as invoked models | +| Tool calls and failures | Sanitized `tool_call` completion events | Attempts with an after event; missing before leaves duration null unless native duration exists | Does not count Agent tool declarations or model selection | +| Sandbox nodes | Deployment administrator node list | Current online, provider-ready, active, retained and capacity fields | A node being online does not prove a Turn can execute | +| Host, database, queue health | No qualified operator projection | Not available | Keep out of Session-scoped charts | Current Runtime history has a 30-second default periodic collection interval and -seven-day retention; the public Web ranges are 1h, 6h, and 24h. Its capability -route distinguishes periodic collection from on-read collection. The Dashboard -must show the source, age, coverage and missing values. An empty interval is not -an observed zero. +seven-day retention; the public Web ranges are 1h, 6h, and 24h. Its table is +renamed in place from `runtime_history_samples` to +`observability_runtime_samples`. Its capability route distinguishes periodic +collection from on-read collection. Missing values and empty intervals are not +observed zeroes. -## Next collection boundary +## Collection boundary and remaining work -Add a separate, authenticated operator metrics service behind Core. Keep the -pinned Agents API resources unchanged. The service should aggregate sanitized -events server-side and expose a bounded read-only extension through +The authenticated operator metrics service runs behind Core. It keeps the pinned +Agents API resources unchanged, aggregates sanitized events server-side, and +exposes a bounded read-only extension through `packages/agents-client`. Browser input may select a fixed range and resolution; it must not select tenants, storage labels, arbitrary PromQL, endpoints or keys. -1. **Ingress:** count completed HTTP requests by route family, method and coarse - outcome (`success`, `client_error`, `server_error`). Record a latency histogram - after the response completes. Exclude health polling or show it separately. - This is transport health, never terminal Turn success. -2. **Execution:** emit one terminal Turn outcome from the durable state transition - winner, with queue wait and run duration measured from persisted timestamps. - Active Turns are a current gauge from Core ownership, not a counter inferred - from sampled CPU. Retries and recovery must not double count terminal outcomes. -3. **Model:** observe actual native model invocation attempts at the harness +1. **Ingress (implemented):** count completed HTTP requests by route family, + method and coarse outcome (`success`, `client_error`, `server_error`). Record a + latency histogram after the response completes. Health polling and the metric + read itself are excluded. The bounded asynchronous queue records drops and + write failures; failed batches are not replayed onto the request path. +2. **Execution (implemented):** query terminal Turns from durable state with queue + wait and run duration measured from persisted timestamps. The query is bounded + to a fixed time window and avoids duplicate terminal events. Active Turns + remain a current Core gauge, not a counter inferred from sampled CPU. +3. **Model (pending):** observe actual native model invocation attempts at the harness boundary, including model identifier, terminal attempt outcome, duration, reported input/output tokens and usage coverage. A saved Agent's configured model is not proof of which model ran. Keep bounded model label cardinality. -4. **Tools:** observe actual tool attempt start and terminal result at the common - Runtime adapter boundary. Use a bounded tool category and outcome label; +4. **Tools (partial):** observe `tool_call` before/after events at the common + execution journal boundary. Use a bounded tool category and outcome label; raw tool names, arguments, output and credentials stay out of metric labels. - Distinguish model tool selection from completed tool execution. -5. **Sandbox:** retain allocation identity and compute generation internally. + Record a completed attempt only after the corresponding event batch has been + persisted. Completion is deduplicated by Turn and call ID. A missing after + event is not counted as a completed attempt. Some harnesses may not emit + these events. +5. **Sandbox (partial):** retain allocation identity and compute generation internally. Extend the normalized sample only after Docker and microsandbox values have matching semantics. Candidates are OOM/exit events, disk usage/limit, network bytes, compute restarts, and allocation/compute startup latency. A missing provider value remains null; lifecycle state remains Core-owned. -6. **Collector health:** count attempted, successful, timed-out and unavailable - samples, queue drops and export failures. Display coverage per interval so - a quiet chart cannot hide collector failure. +6. **Collector health (partial):** request collection writes a minute heartbeat, + including quiet minutes. Request and tool queues report drops and write + failures. Runtime sampler and model-attempt collector health are not yet wired + into this operator projection; an absent row is unavailable, not zero. + +The operator migration adds `observability_request_minute_buckets`, +`observability_model_attempts`, `observability_tool_attempts`, and +`observability_collector_minute_buckets`. Request and collector buckets retain +30 days; model and tool attempts retain seven days. Model attempts remain empty +until a native per-attempt producer is implemented. The deployment administrator +summary supports only 1h, 6h, and 24h and returns no tenant or raw identifiers. Use bounded histograms for p50/p95 latency and rates from counters over complete time buckets. Keep operational cardinality to route family, outcome, provider @@ -65,14 +81,14 @@ must be explicit, with history storage/export optional for execution. ## Dashboard layout and acceptance -The overview keeps loaded Agent/Session counts and current attention. The Runtime -section shows allocation-deduplicated lifecycle, coverage, sample gaps, memory -pressure and existing Live/History trends. A later System section can add request -rate/error rate/p95, terminal Turn outcomes and durations, actual model usage and -tool attempts once their collection is qualified. Every panel needs a source and -freshness label, a coverage denominator, and an unavailable state. It must never -convert missing measurements to zero or equate a configured Sandbox with a -completed execution. +The overview keeps a small current snapshot: Agents, Sessions, active Sandboxes, +reported tokens, measured CPU/memory capacity, and ready nodes. The Observability +tab leads with the four existing Runtime trends, followed by request, Turn, tool, +collector, and node panels. The model panel states that collection is unavailable. The +request rate, error rate, and range latency require complete collector heartbeat +coverage; otherwise they are unavailable. Request route tables show counts only +for covered buckets. Every panel must preserve missing measurements and must not equate +a configured Sandbox with a completed execution. Before shipping those new panels, validate counter deduplication under retries, provider restarts, missing usage, sampler outages and multi-tenant authorization; diff --git a/packages/agents-client/src/index.ts b/packages/agents-client/src/index.ts index 20dd3eaf1..f19125310 100644 --- a/packages/agents-client/src/index.ts +++ b/packages/agents-client/src/index.ts @@ -4,3 +4,4 @@ export { createSSEDecoder } from "./sse"; export type { SSEDecoder, SSEMessage } from "./sse"; export type * from "./types"; export * from "./sandbox-client"; +export * from "./observability-client"; diff --git a/packages/agents-client/src/observability-client.ts b/packages/agents-client/src/observability-client.ts new file mode 100644 index 000000000..1eb4791b4 --- /dev/null +++ b/packages/agents-client/src/observability-client.ts @@ -0,0 +1,42 @@ +import { OpenAIAgentsClient } from "./client"; +import type { ReadOptions } from "./types"; + +export type OperatorMetricsRange = "1h" | "6h" | "24h"; + +export interface RequestMetricBucket { + start: string; + route_family: string; + outcome: "success" | "client_error" | "server_error"; + count: number; + latency_sum_ms: number; + latency_bucket_counts: number[]; +} + +export interface CollectorMetricBucket { + start: string; + source: string; + attempted_count: number; + observed_count: number; + unavailable_count: number; + timeout_count: number; + dropped_count: number; + export_failed_count: number; +} + +export interface OperatorMetricsSummary { + generated_at: string; + start: string; + end: string; + step_seconds: number; + requests: RequestMetricBucket[]; + collector: CollectorMetricBucket[]; + turns: Array<{ start: string; status: "completed" | "failed" | "cancelled"; count: number; queue_p95_ms: number | null; execution_p95_ms: number | null }>; + tools: Array<{ start: string; category: string; outcome: "success" | "error" | "cancelled" | "unknown"; count: number; timed_count: number; duration_p95_ms: number | null }>; +} + +/** Deployment administrator extension; never uses a project credential. */ +export class ObservabilityAdminClient extends OpenAIAgentsClient { + retrieveSummary(range: OperatorMetricsRange, options?: ReadOptions): Promise { + return this.request(`/summary?range=${range}`, { signal: options?.signal }, undefined, false); + } +} diff --git a/services/agents-api/cmd/server/main.go b/services/agents-api/cmd/server/main.go index 985126151..6a75ed261 100644 --- a/services/agents-api/cmd/server/main.go +++ b/services/agents-api/cmd/server/main.go @@ -35,6 +35,7 @@ import ( "github.com/MiniMax-AI-Dev/parsar/internal/obs/log" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/api" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/execution" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtime" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeenrollment" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimehistory" @@ -103,6 +104,9 @@ func run() error { return err } executionStore := store.NewWithCredentialCipherAndOAuthRefresh(pool, credentialKey, oauthClient) + toolMetricsCtx, cancelToolMetrics := context.WithCancel(ctx) + toolMetrics := observability.NewToolRecorder(toolMetricsCtx, executionStore) + defer func() { cancelToolMetrics(); <-toolMetrics.Done() }() if err := executionStore.EnsureProjectScopes(ready, auth.ProjectScopes()); err != nil { return err } @@ -187,6 +191,7 @@ func run() error { } } options = append(options, api.WithProjectAPIKeys(executionStore, keyAdmin), api.WithWriteAudit(executionStore, keyAdmin)) + options = append(options, api.WithOperatorMetrics(executionStore, keyAdmin)) if history.Reader != nil { historyResolver, resolverErr := historystoreresolver.NewResolver(executionStore) if resolverErr != nil { @@ -210,7 +215,7 @@ func run() error { } if registry != nil { dispatcher := &execution.Dispatcher{Store: executionStore, Registry: registry, - ManagedRuntimes: managed, Options: transientOptions} + ManagedRuntimes: managed, Options: transientOptions, ToolRecorder: toolMetrics} worker, err = execution.StartWorker(ctx, dispatcher) if err != nil { @@ -264,6 +269,17 @@ func run() error { startupManaged = managedNodes.setup.selected.Load() } options = append(options, api.WithStartupConfiguration(coreStartupConfiguration(engine, kinds, registry != nil, modelProviderEndpoints, managedRuntimeProviderKind(startupManaged), startupManaged))) + requestMetricsCtx, cancelRequestMetrics := context.WithCancel(ctx) + requestMetrics := observability.NewRequestRecorder(requestMetricsCtx, executionStore) + defer func() { cancelRequestMetrics(); <-requestMetrics.Done() }() + options = append(options, api.WithRequestMetrics(requestMetrics)) + operatorCleanupCtx, cancelOperatorCleanup := context.WithCancel(ctx) + operatorCleanupDone := make(chan struct{}) + go func() { + defer close(operatorCleanupDone) + runOperatorMetricsCleanup(operatorCleanupCtx, executionStore) + }() + defer func() { cancelOperatorCleanup(); <-operatorCleanupDone }() handler, err := api.NewHandler(executionStore, auth, engine, options...) if err != nil { return err diff --git a/services/agents-api/cmd/server/observability_cleanup.go b/services/agents-api/cmd/server/observability_cleanup.go new file mode 100644 index 000000000..ed835f2fc --- /dev/null +++ b/services/agents-api/cmd/server/observability_cleanup.go @@ -0,0 +1,30 @@ +package main + +import ( + "context" + "time" + + "github.com/MiniMax-AI-Dev/parsar/internal/obs/log" +) + +type operatorMetricsPruner interface { + PruneOperatorMetrics(context.Context, time.Time) error +} + +func runOperatorMetricsCleanup(ctx context.Context, pruner operatorMetricsPruner) { + ticker := time.NewTicker(time.Minute) + defer ticker.Stop() + for { + pruneCtx, cancel := context.WithTimeout(ctx, 3*time.Second) + err := pruner.PruneOperatorMetrics(pruneCtx, time.Now().UTC()) + cancel() + if err != nil && ctx.Err() == nil { + log.Ctx(ctx).Warn("Operator metrics retention cleanup failed") + } + select { + case <-ctx.Done(): + return + case <-ticker.C: + } + } +} diff --git a/services/agents-api/internal/api/handler.go b/services/agents-api/internal/api/handler.go index 7cb048e89..4db24b692 100644 --- a/services/agents-api/internal/api/handler.go +++ b/services/agents-api/internal/api/handler.go @@ -13,6 +13,7 @@ import ( "github.com/MiniMax-AI-Dev/parsar/internal/obs/log" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/execution" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/identity" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" "github.com/go-chi/chi/v5" "github.com/go-chi/chi/v5/middleware" @@ -61,6 +62,8 @@ type Handler struct { subagents SubagentStore runtimeObservations RuntimeObservationService runtimeHistory RuntimeHistoryService + requestMetrics *observability.RequestRecorder + operatorMetrics OperatorMetricsStore startup *v1.CoreStartupConfiguration } @@ -92,6 +95,9 @@ func NewHandler(s ResourceStore, auth *Authenticator, engine string, options ... func (h *Handler) routes() *chi.Mux { router := chi.NewRouter() router.Use(agentsResponseHeaders, log.HTTPMiddleware, middleware.GetHead) + if h.requestMetrics != nil { + router.Use(h.recordRequestMetrics) + } router.MethodNotAllowed(methodNotAllowed) router.Get("/healthz", func(w http.ResponseWriter, _ *http.Request) { writeJSON(w, http.StatusOK, map[string]string{"status": "ok"}) @@ -109,6 +115,7 @@ func (h *Handler) routes() *chi.Mux { h.registerSandboxManagerRoutes(router) h.registerProjectAPIKeyRoutes(router) h.registerWriteAuditRoutes(router) + h.registerOperatorMetricsRoutes(router) h.registerEnvironmentExecutorRoutes(router) router.Route("/v1", func(r chi.Router) { r.Use(h.authenticate) diff --git a/services/agents-api/internal/api/observability_read.go b/services/agents-api/internal/api/observability_read.go new file mode 100644 index 000000000..fd79ec262 --- /dev/null +++ b/services/agents-api/internal/api/observability_read.go @@ -0,0 +1,88 @@ +package api + +import ( + "context" + "net/http" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" + "github.com/go-chi/chi/v5" +) + +type OperatorMetricsStore interface { + ReadOperatorMetrics(context.Context, time.Time, time.Time, time.Duration) (store.OperatorMetrics, error) +} + +type OperatorMetricsResponse struct { + GeneratedAt time.Time `json:"generated_at"` + Start time.Time `json:"start"` + End time.Time `json:"end"` + StepSeconds int64 `json:"step_seconds"` + Requests []store.RequestMetricBucket `json:"requests"` + Collector []store.CollectorMetricBucket `json:"collector"` + Turns []store.TurnMetricBucket `json:"turns"` + Tools []store.ToolMetricBucket `json:"tools"` +} + +func WithOperatorMetrics(s OperatorMetricsStore, auth *DeploymentAuthenticator) Option { + return func(h *Handler) { + h.operatorMetrics = s + if auth != nil { + h.deploymentAuth = auth + } + } +} + +func (h *Handler) registerOperatorMetricsRoutes(r chi.Router) { + if h.operatorMetrics == nil || h.deploymentAuth == nil { + return + } + r.Group(func(r chi.Router) { + r.Use(h.deploymentAuth.authenticate) + r.Get("/core/v1/observability/summary", h.getOperatorMetrics) + r.Head("/core/v1/observability/summary", methodNotAllowed) + }) +} + +// @Summary Retrieve bounded operator observability metrics +// @Description Deployment administrator only. Returns fixed-range aggregate requests and collector coverage without paths, payloads or project identifiers. +// @Tags Observability +// @Produce json +// @Security DeploymentAdminAuth +// @Param range query string false "1h, 6h, or 24h (default 1h)" +// @Success 200 {object} api.OperatorMetricsResponse +// @Failure 400,401,500,503 {object} v1.ErrorResponse +// @Router /core/v1/observability/summary [get] +func (h *Handler) getOperatorMetrics(w http.ResponseWriter, r *http.Request) { + values := r.URL.Query() + for key, entries := range values { + if key != "range" || len(entries) != 1 { + writeStoreError(w, r, store.ErrInvalidInput) + return + } + } + span, step := time.Hour, time.Minute + switch values.Get("range") { + case "", "1h": + case "6h": + span, step = 6*time.Hour, 5*time.Minute + case "24h": + span, step = 24*time.Hour, 15*time.Minute + default: + writeStoreError(w, r, store.ErrInvalidInput) + return + } + end := time.Now().UTC().Truncate(step) + start := end.Add(-span) + ctx, cancel := context.WithTimeout(r.Context(), 5*time.Second) + defer cancel() + metrics, err := h.operatorMetrics.ReadOperatorMetrics(ctx, start, end, step) + if err != nil { + writeStoreError(w, r, err) + return + } + writeJSON(w, http.StatusOK, OperatorMetricsResponse{ + GeneratedAt: time.Now().UTC(), Start: start, End: end, StepSeconds: int64(step / time.Second), + Requests: metrics.Requests, Collector: metrics.Collector, Turns: metrics.Turns, Tools: metrics.Tools, + }) +} diff --git a/services/agents-api/internal/api/observability_read_test.go b/services/agents-api/internal/api/observability_read_test.go new file mode 100644 index 000000000..5f9e5ee47 --- /dev/null +++ b/services/agents-api/internal/api/observability_read_test.go @@ -0,0 +1,62 @@ +package api + +import ( + "context" + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" + + "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/device" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" +) + +type operatorMetricsStub struct{ calls int } + +func (s *operatorMetricsStub) ReadOperatorMetrics(_ context.Context, start, end time.Time, step time.Duration) (store.OperatorMetrics, error) { + s.calls++ + if end.Sub(start) != time.Hour || step != time.Minute { + return store.OperatorMetrics{}, store.ErrInvalidInput + } + return store.OperatorMetrics{Requests: []store.RequestMetricBucket{}, Collector: []store.CollectorMetricBucket{}, Turns: []store.TurnMetricBucket{}, Tools: []store.ToolMetricBucket{}}, nil +} + +func TestOperatorMetricsRequireAdministratorAndBoundedRange(t *testing.T) { + project, err := NewAuthenticator([]APIKey{callerBinding()}) + if err != nil { + t.Fatal(err) + } + admin, err := NewDeploymentAuthenticator([]string{device.HashCredential("administrator")}) + if err != nil { + t.Fatal(err) + } + metrics := &operatorMetricsStub{} + h, err := NewHandler(&recordingStore{}, project, "codex", WithOperatorMetrics(metrics, admin)) + if err != nil { + t.Fatal(err) + } + request := func(path, key string) *httptest.ResponseRecorder { + req := httptest.NewRequest(http.MethodGet, path, nil) + req.Header.Set("Authorization", "Bearer "+key) + response := httptest.NewRecorder() + h.ServeHTTP(response, req) + return response + } + for _, key := range []string{"caller", ""} { + response := request("/core/v1/observability/summary?range=1h", key) + if response.Code != http.StatusUnauthorized || metrics.calls != 0 { + t.Fatal("project or missing key reached operator metrics", response.Code, metrics.calls) + } + } + for _, path := range []string{"/core/v1/observability/summary?range=30d", "/core/v1/observability/summary?range=1h&range=24h", "/core/v1/observability/summary?tenant_id=x"} { + response := request(path, "administrator") + if response.Code != http.StatusBadRequest || metrics.calls != 0 { + t.Fatal("unbounded metrics query admitted", response.Code, metrics.calls) + } + } + response := request("/core/v1/observability/summary?range=1h", "administrator") + if response.Code != http.StatusOK || metrics.calls != 1 || !strings.Contains(response.Body.String(), `"requests":[]`) { + t.Fatal("administrator metrics read failed", response.Code, metrics.calls, response.Body.String()) + } +} diff --git a/services/agents-api/internal/api/observability_requests.go b/services/agents-api/internal/api/observability_requests.go new file mode 100644 index 000000000..7c420df60 --- /dev/null +++ b/services/agents-api/internal/api/observability_requests.go @@ -0,0 +1,76 @@ +package api + +import ( + "net/http" + "strings" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" + "github.com/go-chi/chi/v5" + "github.com/go-chi/chi/v5/middleware" +) + +func WithRequestMetrics(recorder *observability.RequestRecorder) Option { + return func(h *Handler) { h.requestMetrics = recorder } +} + +func (h *Handler) recordRequestMetrics(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.URL.Path == "/healthz" || strings.HasPrefix(r.URL.Path, "/core/v1/observability/") { + next.ServeHTTP(w, r) + return + } + start := time.Now() + wrapped := middleware.NewWrapResponseWriter(w, r.ProtoMajor) + next.ServeHTTP(wrapped, r) + status := wrapped.Status() + if status == 0 { + status = http.StatusOK + } + outcome := "success" + if status >= 500 { + outcome = "server_error" + } else if status >= 400 { + outcome = "client_error" + } + method := r.Method + switch method { + case http.MethodGet, http.MethodPost, http.MethodPut, http.MethodPatch, http.MethodDelete, http.MethodHead: + default: + method = "OTHER" + } + pattern := "" + if route := chi.RouteContext(r.Context()); route != nil { + pattern = route.RoutePattern() + } + h.requestMetrics.Record(observability.Request{ + CompletedAt: time.Now().UTC(), RouteFamily: requestRouteFamily(pattern), + Method: method, Outcome: outcome, Latency: time.Since(start), + }) + }) +} + +func requestRouteFamily(pattern string) string { + switch { + case strings.HasPrefix(pattern, "/core/v1/sandbox"): + return "sandbox_admin" + case strings.HasPrefix(pattern, "/core/v1"): + return "core_admin" + case strings.HasPrefix(pattern, "/v1/agents/sessions"): + return "sessions" + case strings.HasPrefix(pattern, "/v1/agents/runtime"): + return "runtime" + case strings.HasPrefix(pattern, "/v1/agents/environments"): + return "environments" + case strings.HasPrefix(pattern, "/v1/agents"): + return "agents" + case strings.HasPrefix(pattern, "/v1/files"): + return "files" + case strings.HasPrefix(pattern, "/v1/vaults"): + return "vaults" + case strings.HasPrefix(pattern, "/v1/skills"): + return "skills" + default: + return "other" + } +} diff --git a/services/agents-api/internal/db/queries/runtime_history.sql b/services/agents-api/internal/db/queries/runtime_history.sql index d0c019024..6a20f0a00 100644 --- a/services/agents-api/internal/db/queries/runtime_history.sql +++ b/services/agents-api/internal/db/queries/runtime_history.sql @@ -1,5 +1,5 @@ -- name: InsertRuntimeHistorySample :exec -INSERT INTO runtime_history_samples ( +INSERT INTO observability_runtime_samples ( tenant_id, session_id, environment_id, resolved_at_ns, allocation_id, provider_type, status, observed_at_ns, started_at_ns, cpu_usage_seconds, cpu_capacity_cores, memory_usage_bytes, memory_limit_bytes, input_tokens, output_tokens @@ -12,7 +12,7 @@ WHERE s.tenant_id = sqlc.arg(tenant_id) AND s.id = sqlc.arg(session_id) AND e.id ON CONFLICT (tenant_id, session_id, environment_id, resolved_at_ns) DO NOTHING; -- name: ListRuntimeHistorySamples :many -SELECT * FROM runtime_history_samples +SELECT * FROM observability_runtime_samples WHERE tenant_id = sqlc.arg(tenant_id) AND session_id = sqlc.arg(session_id) AND environment_id = sqlc.arg(environment_id) AND resolved_at_ns >= sqlc.arg(start_ns) AND resolved_at_ns < sqlc.arg(end_ns) ORDER BY resolved_at_ns @@ -20,10 +20,10 @@ LIMIT sqlc.arg(row_limit); -- name: PruneRuntimeHistorySamples :execrows WITH expired AS ( - SELECT p.tenant_id, p.session_id, p.environment_id, p.resolved_at_ns FROM runtime_history_samples p + SELECT p.tenant_id, p.session_id, p.environment_id, p.resolved_at_ns FROM observability_runtime_samples p WHERE p.resolved_at_ns < sqlc.arg(before_ns) ORDER BY p.resolved_at_ns LIMIT 256 FOR UPDATE SKIP LOCKED ) -DELETE FROM runtime_history_samples h USING expired e +DELETE FROM observability_runtime_samples h USING expired e WHERE h.tenant_id = e.tenant_id AND h.session_id = e.session_id AND h.environment_id = e.environment_id AND h.resolved_at_ns = e.resolved_at_ns; diff --git a/services/agents-api/internal/db/sqlc/models.go b/services/agents-api/internal/db/sqlc/models.go index 0c64280b3..3cf0776df 100644 --- a/services/agents-api/internal/db/sqlc/models.go +++ b/services/agents-api/internal/db/sqlc/models.go @@ -136,6 +136,73 @@ type InitialEnvironmentFile struct { Contents []byte `json:"contents"` } +type ObservabilityCollectorMinuteBucket struct { + BucketStart pgtype.Timestamptz `json:"bucket_start"` + Source string `json:"source"` + AttemptedCount int64 `json:"attempted_count"` + ObservedCount int64 `json:"observed_count"` + UnavailableCount int64 `json:"unavailable_count"` + TimeoutCount int64 `json:"timeout_count"` + DroppedCount int64 `json:"dropped_count"` + ExportFailedCount int64 `json:"export_failed_count"` +} + +type ObservabilityModelAttempt struct { + ID pgtype.UUID `json:"id"` + TenantID pgtype.UUID `json:"tenant_id"` + SessionID pgtype.UUID `json:"session_id"` + TurnID pgtype.UUID `json:"turn_id"` + StartedAt pgtype.Timestamptz `json:"started_at"` + FinishedAt pgtype.Timestamptz `json:"finished_at"` + ModelFamily string `json:"model_family"` + ProviderType string `json:"provider_type"` + Outcome string `json:"outcome"` + DurationMs int64 `json:"duration_ms"` + InputTokens pgtype.Int8 `json:"input_tokens"` + OutputTokens pgtype.Int8 `json:"output_tokens"` + UsageStatus string `json:"usage_status"` +} + +type ObservabilityRequestMinuteBucket struct { + BucketStart pgtype.Timestamptz `json:"bucket_start"` + RouteFamily string `json:"route_family"` + Method string `json:"method"` + Outcome string `json:"outcome"` + RequestCount int64 `json:"request_count"` + LatencySumMs int64 `json:"latency_sum_ms"` + LatencyBucketCounts []int64 `json:"latency_bucket_counts"` +} + +type ObservabilityRuntimeSample struct { + TenantID pgtype.UUID `json:"tenant_id"` + SessionID pgtype.UUID `json:"session_id"` + EnvironmentID pgtype.UUID `json:"environment_id"` + ResolvedAtNs int64 `json:"resolved_at_ns"` + AllocationID pgtype.UUID `json:"allocation_id"` + ProviderType string `json:"provider_type"` + Status string `json:"status"` + ObservedAtNs pgtype.Int8 `json:"observed_at_ns"` + StartedAtNs pgtype.Int8 `json:"started_at_ns"` + CpuUsageSeconds pgtype.Float8 `json:"cpu_usage_seconds"` + CpuCapacityCores pgtype.Float8 `json:"cpu_capacity_cores"` + MemoryUsageBytes pgtype.Int8 `json:"memory_usage_bytes"` + MemoryLimitBytes pgtype.Int8 `json:"memory_limit_bytes"` + InputTokens pgtype.Int8 `json:"input_tokens"` + OutputTokens pgtype.Int8 `json:"output_tokens"` +} + +type ObservabilityToolAttempt struct { + ID pgtype.UUID `json:"id"` + TenantID pgtype.UUID `json:"tenant_id"` + SessionID pgtype.UUID `json:"session_id"` + TurnID pgtype.UUID `json:"turn_id"` + StartedAt pgtype.Timestamptz `json:"started_at"` + FinishedAt pgtype.Timestamptz `json:"finished_at"` + ToolCategory string `json:"tool_category"` + Outcome string `json:"outcome"` + DurationMs pgtype.Int8 `json:"duration_ms"` +} + type ProjectApiKey struct { ID pgtype.UUID `json:"id"` Name string `json:"name"` @@ -213,24 +280,6 @@ type RuntimeDeviceAuthority struct { CredentialHash string `json:"credential_hash"` } -type RuntimeHistorySample struct { - TenantID pgtype.UUID `json:"tenant_id"` - SessionID pgtype.UUID `json:"session_id"` - EnvironmentID pgtype.UUID `json:"environment_id"` - ResolvedAtNs int64 `json:"resolved_at_ns"` - AllocationID pgtype.UUID `json:"allocation_id"` - ProviderType string `json:"provider_type"` - Status string `json:"status"` - ObservedAtNs pgtype.Int8 `json:"observed_at_ns"` - StartedAtNs pgtype.Int8 `json:"started_at_ns"` - CpuUsageSeconds pgtype.Float8 `json:"cpu_usage_seconds"` - CpuCapacityCores pgtype.Float8 `json:"cpu_capacity_cores"` - MemoryUsageBytes pgtype.Int8 `json:"memory_usage_bytes"` - MemoryLimitBytes pgtype.Int8 `json:"memory_limit_bytes"` - InputTokens pgtype.Int8 `json:"input_tokens"` - OutputTokens pgtype.Int8 `json:"output_tokens"` -} - type RuntimeNode struct { ID pgtype.UUID `json:"id"` InstallationID pgtype.UUID `json:"installation_id"` diff --git a/services/agents-api/internal/db/sqlc/runtime_history.sql.go b/services/agents-api/internal/db/sqlc/runtime_history.sql.go index 1c6a84eac..24567a81c 100644 --- a/services/agents-api/internal/db/sqlc/runtime_history.sql.go +++ b/services/agents-api/internal/db/sqlc/runtime_history.sql.go @@ -12,7 +12,7 @@ import ( ) const insertRuntimeHistorySample = `-- name: InsertRuntimeHistorySample :exec -INSERT INTO runtime_history_samples ( +INSERT INTO observability_runtime_samples ( tenant_id, session_id, environment_id, resolved_at_ns, allocation_id, provider_type, status, observed_at_ns, started_at_ns, cpu_usage_seconds, cpu_capacity_cores, memory_usage_bytes, memory_limit_bytes, input_tokens, output_tokens @@ -65,7 +65,7 @@ func (q *Queries) InsertRuntimeHistorySample(ctx context.Context, arg InsertRunt } const listRuntimeHistorySamples = `-- name: ListRuntimeHistorySamples :many -SELECT tenant_id, session_id, environment_id, resolved_at_ns, allocation_id, provider_type, status, observed_at_ns, started_at_ns, cpu_usage_seconds, cpu_capacity_cores, memory_usage_bytes, memory_limit_bytes, input_tokens, output_tokens FROM runtime_history_samples +SELECT tenant_id, session_id, environment_id, resolved_at_ns, allocation_id, provider_type, status, observed_at_ns, started_at_ns, cpu_usage_seconds, cpu_capacity_cores, memory_usage_bytes, memory_limit_bytes, input_tokens, output_tokens FROM observability_runtime_samples WHERE tenant_id = $1 AND session_id = $2 AND environment_id = $3 AND resolved_at_ns >= $4 AND resolved_at_ns < $5 ORDER BY resolved_at_ns @@ -81,7 +81,7 @@ type ListRuntimeHistorySamplesParams struct { RowLimit int32 `json:"row_limit"` } -func (q *Queries) ListRuntimeHistorySamples(ctx context.Context, arg ListRuntimeHistorySamplesParams) ([]RuntimeHistorySample, error) { +func (q *Queries) ListRuntimeHistorySamples(ctx context.Context, arg ListRuntimeHistorySamplesParams) ([]ObservabilityRuntimeSample, error) { rows, err := q.db.Query(ctx, listRuntimeHistorySamples, arg.TenantID, arg.SessionID, @@ -94,9 +94,9 @@ func (q *Queries) ListRuntimeHistorySamples(ctx context.Context, arg ListRuntime return nil, err } defer rows.Close() - items := []RuntimeHistorySample{} + items := []ObservabilityRuntimeSample{} for rows.Next() { - var i RuntimeHistorySample + var i ObservabilityRuntimeSample if err := rows.Scan( &i.TenantID, &i.SessionID, @@ -126,11 +126,11 @@ func (q *Queries) ListRuntimeHistorySamples(ctx context.Context, arg ListRuntime const pruneRuntimeHistorySamples = `-- name: PruneRuntimeHistorySamples :execrows WITH expired AS ( - SELECT p.tenant_id, p.session_id, p.environment_id, p.resolved_at_ns FROM runtime_history_samples p + SELECT p.tenant_id, p.session_id, p.environment_id, p.resolved_at_ns FROM observability_runtime_samples p WHERE p.resolved_at_ns < $1 ORDER BY p.resolved_at_ns LIMIT 256 FOR UPDATE SKIP LOCKED ) -DELETE FROM runtime_history_samples h USING expired e +DELETE FROM observability_runtime_samples h USING expired e WHERE h.tenant_id = e.tenant_id AND h.session_id = e.session_id AND h.environment_id = e.environment_id AND h.resolved_at_ns = e.resolved_at_ns ` diff --git a/services/agents-api/internal/execution/delivery.go b/services/agents-api/internal/execution/delivery.go index 52b6d59dd..9e97f14c3 100644 --- a/services/agents-api/internal/execution/delivery.go +++ b/services/agents-api/internal/execution/delivery.go @@ -65,7 +65,7 @@ func (d *Dispatcher) deliver(ctx context.Context, tenantID, sessionID string, pe } }() journal := &journal{store: d.Store, tenant: tenantID, session: sessionID, turn: request.RunID, next: 1, - observeSubagents: request.ObserveSubagentIdentities} + observeSubagents: request.ObserveSubagentIdentities, toolRecorder: d.ToolRecorder} defer func() { finishCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second) defer cancel() diff --git a/services/agents-api/internal/execution/dispatcher.go b/services/agents-api/internal/execution/dispatcher.go index 3d9c98e03..4370da884 100644 --- a/services/agents-api/internal/execution/dispatcher.go +++ b/services/agents-api/internal/execution/dispatcher.go @@ -11,6 +11,7 @@ import ( v1 "github.com/MiniMax-AI-Dev/parsar/contracts/agents-api/v1" "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/gateway" "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/proto" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" ) @@ -36,6 +37,10 @@ type Dispatcher struct { Options func(context.Context, store.Session) (map[string]any, error) // ManagedRuntimes is optional internal provisioning; it does not admit hosted API requests. ManagedRuntimes *RuntimeProvider + // ToolRecorder receives sanitized terminal observations without affecting execution. + ToolRecorder interface { + Record(observability.ToolAttempt) + } } type Result struct { diff --git a/services/agents-api/internal/execution/journal.go b/services/agents-api/internal/execution/journal.go index ae22830de..0835a42c2 100644 --- a/services/agents-api/internal/execution/journal.go +++ b/services/agents-api/internal/execution/journal.go @@ -6,7 +6,9 @@ import ( "time" "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/proto" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" + "github.com/google/uuid" ) type journal struct { @@ -14,9 +16,14 @@ type journal struct { tenant, session, turn string next int32 batch []store.ExecutionEvent + batchTimes []time.Time bytes int pendingCount int observeSubagents bool + toolRecorder interface { + Record(observability.ToolAttempt) + } + toolStarts map[string]time.Time } type eventWriter interface { @@ -67,10 +74,66 @@ func (j *journal) enqueue(env proto.Envelope) error { return store.ErrEventLimit } j.batch = append(j.batch, store.ExecutionEvent{Kind: env.Type, Payload: env.Payload}) + j.batchTimes = append(j.batchTimes, time.Now().UTC()) j.bytes += len(env.Payload) return nil } +func (j *journal) observeToolAttempt(raw json.RawMessage, observedAt time.Time) { + if j.toolRecorder == nil { + return + } + var call proto.ToolCallPayload + if json.Unmarshal(raw, &call) != nil || call.ID == "" { + return + } + if call.Stage == "before" { + if j.toolStarts == nil { + j.toolStarts = make(map[string]time.Time) + } + if _, exists := j.toolStarts[call.ID]; !exists { + j.toolStarts[call.ID] = observedAt + } + return + } + if call.Stage != "after" { + return + } + finished := observedAt + var started *time.Time + if value, exists := j.toolStarts[call.ID]; exists { + started = &value + delete(j.toolStarts, call.ID) + } + category, outcome := "other", "unknown" + var duration *int64 + if call.Observation != nil { + switch call.Observation.Kind { + case "command", "mcp", "function", "web_search", "file": + category = call.Observation.Kind + } + switch call.Observation.Status { + case "completed": + outcome = "success" + case "failed": + outcome = "error" + } + if call.Observation.DurationMS != nil && *call.Observation.DurationMS >= 0 { + value := *call.Observation.DurationMS + duration = &value + } + } + if duration == nil && started != nil { + value := finished.Sub(*started).Milliseconds() + duration = &value + } + j.toolRecorder.Record(observability.ToolAttempt{ + ID: uuid.NewSHA1(uuid.NameSpaceOID, []byte(j.turn+":"+call.ID)).String(), + TenantID: j.tenant, SessionID: j.session, TurnID: j.turn, + StartedAt: started, FinishedAt: finished, Category: category, Outcome: outcome, DurationMS: duration, + }) +} + func (j *journal) flush(ctx context.Context) error { if len(j.batch) == 0 { return nil @@ -96,12 +159,19 @@ func (j *journal) flush(ctx context.Context) error { if err := j.store.AppendTurnEvents(ctx, j.tenant, j.session, j.turn, j.next, j.batch[:count]); err != nil { return err } + for index, event := range j.batch[:count] { + if event.Kind == proto.TypeToolCall { + j.observeToolAttempt(event.Payload, j.batchTimes[index]) + } + } j.pendingCount = 0 j.next += int32(count) j.bytes -= size j.batch = j.batch[count:] + j.batchTimes = j.batchTimes[count:] } j.batch = nil + j.batchTimes = nil return nil } diff --git a/services/agents-api/internal/execution/observability_tools_test.go b/services/agents-api/internal/execution/observability_tools_test.go new file mode 100644 index 000000000..2a4cabf28 --- /dev/null +++ b/services/agents-api/internal/execution/observability_tools_test.go @@ -0,0 +1,60 @@ +package execution + +import ( + "context" + "errors" + "testing" + + "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/proto" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" +) + +type toolCapture struct{ attempts []observability.ToolAttempt } + +func (c *toolCapture) Record(value observability.ToolAttempt) { c.attempts = append(c.attempts, value) } + +type toolEventWriter struct{ err error } + +func (w *toolEventWriter) AppendTurnEvents(_ context.Context, _, _, _ string, _ int32, _ []store.ExecutionEvent) error { + return w.err +} + +func TestToolAttemptProjectsOnlyTerminalSanitizedEvidence(t *testing.T) { + capture := &toolCapture{} + writer := &toolEventWriter{} + j := &journal{store: writer, tenant: "tenant", session: "session", turn: "turn", toolRecorder: capture} + before, err := proto.NewEnvelope(proto.TypeToolCall, "turn", proto.ToolCallPayload{ + ID: "native-call", Stage: "before", Name: "secret-tool-name", + Observation: &proto.ToolObservation{Kind: "command", Status: "in_progress", Command: "secret command"}, + }) + if err != nil || j.enqueue(before) != nil || len(capture.attempts) != 0 { + t.Fatalf("tool start should not count as terminal: %v", err) + } + after, err := proto.NewEnvelope(proto.TypeToolCall, "turn", proto.ToolCallPayload{ + ID: "native-call", Stage: "after", Name: "secret-tool-name", + Observation: &proto.ToolObservation{Kind: "command", Status: "failed", Command: "secret command"}, + }) + if err != nil || j.enqueue(after) != nil || len(capture.attempts) != 0 { + t.Fatalf("tool attempt must wait for durable event append: %v", err) + } + writer.err = errors.New("event persistence failed") + if j.flush(context.Background()) == nil || len(capture.attempts) != 0 { + t.Fatal("failed event append produced a tool metric") + } + writer.err = nil + if err := j.flush(context.Background()); err != nil || len(capture.attempts) != 1 { + t.Fatalf("durable terminal tool observation missing: %v", err) + } + got := capture.attempts[0] + if got.Category != "command" || got.Outcome != "error" || got.StartedAt == nil || got.DurationMS == nil || + got.ID == "native-call" || got.ID == "secret-tool-name" || got.TurnID != "turn" { + t.Fatalf("tool attempt lost its bounded projection: %+v", got) + } + if err := j.enqueue(after); err != nil { + t.Fatal(err) + } + if err := j.flush(context.Background()); err != nil || capture.attempts[1].ID != got.ID { + t.Fatalf("retry did not retain deterministic attempt identity: %v", err) + } +} diff --git a/services/agents-api/internal/observability/requests.go b/services/agents-api/internal/observability/requests.go new file mode 100644 index 000000000..d871906de --- /dev/null +++ b/services/agents-api/internal/observability/requests.go @@ -0,0 +1,134 @@ +// Package observability collects bounded, side-channel operator metrics. +package observability + +import ( + "context" + "sync/atomic" + "time" +) + +// Request contains only bounded labels; it never carries a URL, principal, or payload. +type Request struct { + CompletedAt time.Time + RouteFamily string + Method string + Outcome string + Latency time.Duration +} + +type RequestBucket struct { + Start time.Time + RouteFamily string + Method string + Outcome string + Count int64 + LatencySumMS int64 + LatencyCounts [11]int64 +} + +type RequestSink interface { + WriteRequestBuckets(context.Context, []RequestBucket, int64, int64) error +} + +// RequestRecorder batches metrics away from the HTTP response path. +type RequestRecorder struct { + queue chan Request + sink RequestSink + dropped atomic.Int64 + failed atomic.Int64 + done chan struct{} +} + +func NewRequestRecorder(ctx context.Context, sink RequestSink) *RequestRecorder { + r := &RequestRecorder{queue: make(chan Request, 2048), sink: sink, done: make(chan struct{})} + go r.run(ctx) + return r +} + +func (r *RequestRecorder) Record(value Request) { + select { + case r.queue <- value: + default: + r.dropped.Add(1) + } +} + +func (r *RequestRecorder) Done() <-chan struct{} { return r.done } + +func (r *RequestRecorder) run(ctx context.Context) { + defer close(r.done) + ticker := time.NewTicker(time.Second) + defer ticker.Stop() + pending := make(map[RequestBucket]*RequestBucket) + var lastHeartbeat time.Time + flush := func() { + buckets := make([]RequestBucket, 0, len(pending)) + var pendingCount int64 + for _, value := range pending { + buckets = append(buckets, *value) + pendingCount += value.Count + } + dropped := r.dropped.Swap(0) + failed := r.failed.Swap(0) + minute := time.Now().UTC().Truncate(time.Minute) + if len(buckets) == 0 && dropped == 0 && failed == 0 && minute.Equal(lastHeartbeat) { + return + } + writeCtx, cancel := context.WithTimeout(context.Background(), 2*time.Second) + err := r.sink.WriteRequestBuckets(writeCtx, buckets, dropped, failed) + cancel() + if err != nil { + r.failed.Add(1) + r.dropped.Add(dropped + pendingCount) + } else { + lastHeartbeat = minute + } + clear(pending) + } + add := func(value Request) { + if value.CompletedAt.IsZero() { + return + } + start := value.CompletedAt.UTC().Truncate(time.Minute) + key := RequestBucket{Start: start, RouteFamily: value.RouteFamily, Method: value.Method, Outcome: value.Outcome} + bucket := pending[key] + if bucket == nil { + bucket = &key + pending[key] = bucket + } + bucket.Count++ + ms := value.Latency.Milliseconds() + if ms < 0 { + ms = 0 + } + bucket.LatencySumMS += ms + bucket.LatencyCounts[latencyIndex(ms)]++ + } + for { + select { + case value := <-r.queue: + add(value) + case <-ticker.C: + flush() + case <-ctx.Done(): + for { + select { + case value := <-r.queue: + add(value) + default: + flush() + return + } + } + } + } +} + +func latencyIndex(ms int64) int { + for i, bound := range [...]int64{10, 25, 50, 100, 250, 500, 1000, 2500, 5000, 10000} { + if ms <= bound { + return i + } + } + return 10 +} diff --git a/services/agents-api/internal/observability/requests_test.go b/services/agents-api/internal/observability/requests_test.go new file mode 100644 index 000000000..c6af0950f --- /dev/null +++ b/services/agents-api/internal/observability/requests_test.go @@ -0,0 +1,40 @@ +package observability + +import ( + "context" + "testing" + "time" +) + +type capturedRequests struct { + buckets []RequestBucket + dropped int64 +} + +func (s *capturedRequests) WriteRequestBuckets(_ context.Context, buckets []RequestBucket, dropped, _ int64) error { + s.buckets = append(s.buckets, buckets...) + s.dropped += dropped + return nil +} + +func TestRequestRecorderAggregatesBoundedLatencyBuckets(t *testing.T) { + ctx, cancel := context.WithCancel(context.Background()) + sink := &capturedRequests{} + recorder := NewRequestRecorder(ctx, sink) + at := time.Date(2026, 9, 25, 10, 20, 30, 0, time.UTC) + recorder.Record(Request{CompletedAt: at, RouteFamily: "sessions", Method: "GET", Outcome: "success", Latency: 50 * time.Millisecond}) + recorder.Record(Request{CompletedAt: at.Add(3 * time.Second), RouteFamily: "sessions", Method: "GET", Outcome: "success", Latency: 11 * time.Second}) + cancel() + select { + case <-recorder.Done(): + case <-time.After(3 * time.Second): + t.Fatal("request recorder did not drain on shutdown") + } + if len(sink.buckets) != 1 { + t.Fatalf("expected one aggregate bucket, got %d", len(sink.buckets)) + } + bucket := sink.buckets[0] + if bucket.Count != 2 || bucket.Start != at.Truncate(time.Minute) || bucket.LatencySumMS != 11050 || bucket.LatencyCounts[2] != 1 || bucket.LatencyCounts[10] != 1 || sink.dropped != 0 { + t.Fatalf("unexpected request aggregate: %+v, dropped=%d", bucket, sink.dropped) + } +} diff --git a/services/agents-api/internal/observability/tools.go b/services/agents-api/internal/observability/tools.go new file mode 100644 index 000000000..ab4d7a34f --- /dev/null +++ b/services/agents-api/internal/observability/tools.go @@ -0,0 +1,90 @@ +package observability + +import ( + "context" + "sync/atomic" + "time" +) + +// ToolAttempt is a sanitized terminal execution observation. +type ToolAttempt struct { + ID string + TenantID string + SessionID string + TurnID string + StartedAt *time.Time + FinishedAt time.Time + Category string + Outcome string + DurationMS *int64 +} + +type ToolSink interface { + WriteToolAttempts(context.Context, []ToolAttempt, int64, int64) error +} + +type ToolRecorder struct { + queue chan ToolAttempt + sink ToolSink + dropped atomic.Int64 + failed atomic.Int64 + done chan struct{} +} + +func NewToolRecorder(ctx context.Context, sink ToolSink) *ToolRecorder { + r := &ToolRecorder{queue: make(chan ToolAttempt, 1024), sink: sink, done: make(chan struct{})} + go r.run(ctx) + return r +} + +func (r *ToolRecorder) Record(value ToolAttempt) { + select { + case r.queue <- value: + default: + r.dropped.Add(1) + } +} + +func (r *ToolRecorder) Done() <-chan struct{} { return r.done } + +func (r *ToolRecorder) run(ctx context.Context) { + defer close(r.done) + ticker := time.NewTicker(time.Second) + defer ticker.Stop() + pending := make([]ToolAttempt, 0, 128) + flush := func() { + dropped, failed := r.dropped.Swap(0), r.failed.Swap(0) + if len(pending) == 0 && dropped == 0 && failed == 0 { + return + } + writeCtx, cancel := context.WithTimeout(context.Background(), 2*time.Second) + err := r.sink.WriteToolAttempts(writeCtx, pending, dropped, failed) + cancel() + if err != nil { + r.failed.Add(1) + r.dropped.Add(dropped + int64(len(pending))) + } + pending = pending[:0] + } + for { + select { + case value := <-r.queue: + pending = append(pending, value) + if len(pending) >= 128 { + flush() + } + case <-ticker.C: + flush() + case <-ctx.Done(): + for { + select { + case value := <-r.queue: + pending = append(pending, value) + default: + flush() + return + } + } + } + } +} diff --git a/services/agents-api/internal/store/observability_read.go b/services/agents-api/internal/store/observability_read.go new file mode 100644 index 000000000..211a0835d --- /dev/null +++ b/services/agents-api/internal/store/observability_read.go @@ -0,0 +1,164 @@ +package store + +import ( + "context" + "time" +) + +type RequestMetricBucket struct { + Start time.Time `json:"start"` + RouteFamily string `json:"route_family"` + Outcome string `json:"outcome"` + Count int64 `json:"count"` + LatencySumMS int64 `json:"latency_sum_ms"` + LatencyCounts []int64 `json:"latency_bucket_counts"` +} + +type CollectorMetricBucket struct { + Start time.Time `json:"start"` + Source string `json:"source"` + AttemptedCount int64 `json:"attempted_count"` + ObservedCount int64 `json:"observed_count"` + UnavailableCount int64 `json:"unavailable_count"` + TimeoutCount int64 `json:"timeout_count"` + DroppedCount int64 `json:"dropped_count"` + ExportFailedCount int64 `json:"export_failed_count"` +} + +type OperatorMetrics struct { + Requests []RequestMetricBucket `json:"requests"` + Collector []CollectorMetricBucket `json:"collector"` + Turns []TurnMetricBucket `json:"turns"` + Tools []ToolMetricBucket `json:"tools"` +} + +type ToolMetricBucket struct { + Start time.Time `json:"start"` + Category string `json:"category"` + Outcome string `json:"outcome"` + Count int64 `json:"count"` + TimedCount int64 `json:"timed_count"` + DurationP95MS *float64 `json:"duration_p95_ms"` +} + +type TurnMetricBucket struct { + Start time.Time `json:"start"` + Status string `json:"status"` + Count int64 `json:"count"` + QueueP95MS *float64 `json:"queue_p95_ms"` + ExecutionP95MS *float64 `json:"execution_p95_ms"` +} + +const requestMetricSelect = `SELECT + to_timestamp(floor(extract(epoch FROM bucket_start) / $3) * $3) AS bucket, + route_family, outcome, sum(request_count)::bigint, sum(latency_sum_ms)::bigint, + ARRAY[ + sum(latency_bucket_counts[1])::bigint, sum(latency_bucket_counts[2])::bigint, + sum(latency_bucket_counts[3])::bigint, sum(latency_bucket_counts[4])::bigint, + sum(latency_bucket_counts[5])::bigint, sum(latency_bucket_counts[6])::bigint, + sum(latency_bucket_counts[7])::bigint, sum(latency_bucket_counts[8])::bigint, + sum(latency_bucket_counts[9])::bigint, sum(latency_bucket_counts[10])::bigint, + sum(latency_bucket_counts[11])::bigint + ] AS latency_bucket_counts +FROM observability_request_minute_buckets +WHERE bucket_start >= $1 AND bucket_start < $2 +GROUP BY bucket, route_family, outcome ORDER BY bucket, route_family, outcome` + +const collectorMetricSelect = `SELECT + to_timestamp(floor(extract(epoch FROM bucket_start) / $3) * $3) AS bucket, + source, sum(attempted_count)::bigint, sum(observed_count)::bigint, + sum(unavailable_count)::bigint, sum(timeout_count)::bigint, + sum(dropped_count)::bigint, sum(export_failed_count)::bigint +FROM observability_collector_minute_buckets +WHERE bucket_start >= $1 AND bucket_start < $2 +GROUP BY bucket, source ORDER BY bucket, source` + +const turnMetricSelect = `SELECT + to_timestamp(floor(extract(epoch FROM completed_at) / $3) * $3) AS bucket, + status, count(*)::bigint, + (percentile_disc(0.95) WITHIN GROUP (ORDER BY greatest(extract(epoch FROM started_at - created_at) * 1000, 0)) + FILTER (WHERE started_at IS NOT NULL))::double precision, + (percentile_disc(0.95) WITHIN GROUP (ORDER BY greatest(extract(epoch FROM completed_at - started_at) * 1000, 0)) + FILTER (WHERE started_at IS NOT NULL))::double precision +FROM turns +WHERE completed_at >= $1 AND completed_at < $2 + AND status IN ('completed', 'failed', 'cancelled') +GROUP BY bucket, status ORDER BY bucket, status` + +const toolMetricSelect = `SELECT + to_timestamp(floor(extract(epoch FROM finished_at) / $3) * $3) AS bucket, + tool_category, outcome, count(*)::bigint, count(duration_ms)::bigint, + (percentile_disc(0.95) WITHIN GROUP (ORDER BY duration_ms) + FILTER (WHERE duration_ms IS NOT NULL))::double precision +FROM observability_tool_attempts +WHERE finished_at >= $1 AND finished_at < $2 +GROUP BY bucket, tool_category, outcome ORDER BY bucket, tool_category, outcome` + +// ReadOperatorMetrics returns only bounded labels and aggregate measurements. +func (s *Store) ReadOperatorMetrics(ctx context.Context, start, end time.Time, step time.Duration) (OperatorMetrics, error) { + result := OperatorMetrics{Requests: []RequestMetricBucket{}, Collector: []CollectorMetricBucket{}, Turns: []TurnMetricBucket{}, Tools: []ToolMetricBucket{}} + seconds := int64(step / time.Second) + rows, err := s.pool.Query(ctx, requestMetricSelect, start, end, seconds) + if err != nil { + return result, err + } + for rows.Next() { + var row RequestMetricBucket + if err := rows.Scan(&row.Start, &row.RouteFamily, &row.Outcome, &row.Count, &row.LatencySumMS, &row.LatencyCounts); err != nil { + rows.Close() + return result, err + } + result.Requests = append(result.Requests, row) + } + err = rows.Err() + rows.Close() + if err != nil { + return result, err + } + rows, err = s.pool.Query(ctx, collectorMetricSelect, start, end, seconds) + if err != nil { + return result, err + } + defer rows.Close() + for rows.Next() { + var row CollectorMetricBucket + if err := rows.Scan(&row.Start, &row.Source, &row.AttemptedCount, &row.ObservedCount, &row.UnavailableCount, + &row.TimeoutCount, &row.DroppedCount, &row.ExportFailedCount); err != nil { + return result, err + } + result.Collector = append(result.Collector, row) + } + if err := rows.Err(); err != nil { + return result, err + } + rows.Close() + rows, err = s.pool.Query(ctx, turnMetricSelect, start, end, seconds) + if err != nil { + return result, err + } + defer rows.Close() + for rows.Next() { + var row TurnMetricBucket + if err := rows.Scan(&row.Start, &row.Status, &row.Count, &row.QueueP95MS, &row.ExecutionP95MS); err != nil { + return result, err + } + result.Turns = append(result.Turns, row) + } + if err := rows.Err(); err != nil { + return result, err + } + rows.Close() + rows, err = s.pool.Query(ctx, toolMetricSelect, start, end, seconds) + if err != nil { + return result, err + } + defer rows.Close() + for rows.Next() { + var row ToolMetricBucket + if err := rows.Scan(&row.Start, &row.Category, &row.Outcome, &row.Count, &row.TimedCount, &row.DurationP95MS); err != nil { + return result, err + } + result.Tools = append(result.Tools, row) + } + return result, rows.Err() +} diff --git a/services/agents-api/internal/store/observability_requests.go b/services/agents-api/internal/store/observability_requests.go new file mode 100644 index 000000000..eebff6722 --- /dev/null +++ b/services/agents-api/internal/store/observability_requests.go @@ -0,0 +1,63 @@ +package store + +import ( + "context" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" +) + +const upsertRequestBucket = `INSERT INTO observability_request_minute_buckets AS current + (bucket_start, route_family, method, outcome, request_count, latency_sum_ms, latency_bucket_counts) +VALUES ($1, $2, $3, $4, $5, $6, $7) +ON CONFLICT (bucket_start, route_family, method, outcome) DO UPDATE SET + request_count = current.request_count + EXCLUDED.request_count, + latency_sum_ms = current.latency_sum_ms + EXCLUDED.latency_sum_ms, + latency_bucket_counts = ARRAY[ + current.latency_bucket_counts[1] + EXCLUDED.latency_bucket_counts[1], + current.latency_bucket_counts[2] + EXCLUDED.latency_bucket_counts[2], + current.latency_bucket_counts[3] + EXCLUDED.latency_bucket_counts[3], + current.latency_bucket_counts[4] + EXCLUDED.latency_bucket_counts[4], + current.latency_bucket_counts[5] + EXCLUDED.latency_bucket_counts[5], + current.latency_bucket_counts[6] + EXCLUDED.latency_bucket_counts[6], + current.latency_bucket_counts[7] + EXCLUDED.latency_bucket_counts[7], + current.latency_bucket_counts[8] + EXCLUDED.latency_bucket_counts[8], + current.latency_bucket_counts[9] + EXCLUDED.latency_bucket_counts[9], + current.latency_bucket_counts[10] + EXCLUDED.latency_bucket_counts[10], + current.latency_bucket_counts[11] + EXCLUDED.latency_bucket_counts[11] + ]` + +// WriteRequestBuckets persists asynchronous HTTP aggregates and collector coverage +// together. It is never called by the request handler's response path. +func (s *Store) WriteRequestBuckets(ctx context.Context, buckets []observability.RequestBucket, dropped, failed int64) error { + tx, err := s.pool.Begin(ctx) + if err != nil { + return err + } + defer tx.Rollback(ctx) + var written int64 + for _, bucket := range buckets { + if bucket.Count <= 0 { + continue + } + _, err = tx.Exec(ctx, upsertRequestBucket, bucket.Start, bucket.RouteFamily, bucket.Method, bucket.Outcome, + bucket.Count, bucket.LatencySumMS, bucket.LatencyCounts[:]) + if err != nil { + return err + } + written += bucket.Count + } + _, err = tx.Exec(ctx, `INSERT INTO observability_collector_minute_buckets AS current + (bucket_start, source, attempted_count, observed_count, dropped_count, export_failed_count) + VALUES ($1, 'request', $2, $3, $4, $5) + ON CONFLICT (bucket_start, source) DO UPDATE SET + attempted_count = current.attempted_count + EXCLUDED.attempted_count, + observed_count = current.observed_count + EXCLUDED.observed_count, + dropped_count = current.dropped_count + EXCLUDED.dropped_count, + export_failed_count = current.export_failed_count + EXCLUDED.export_failed_count`, + time.Now().UTC().Truncate(time.Minute), written+dropped, written, dropped, failed) + if err != nil { + return err + } + return tx.Commit(ctx) +} diff --git a/services/agents-api/internal/store/observability_requests_test.go b/services/agents-api/internal/store/observability_requests_test.go new file mode 100644 index 000000000..a886a65ba --- /dev/null +++ b/services/agents-api/internal/store/observability_requests_test.go @@ -0,0 +1,37 @@ +package store + +import ( + "testing" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" +) + +func TestPostgresOperatorRequestBuckets(t *testing.T) { + s, _ := testStore(t) + start := time.Now().UTC().Truncate(time.Minute).Add(-time.Minute) + first := observability.RequestBucket{Start: start, RouteFamily: "sessions", Method: "GET", Outcome: "success", Count: 2, LatencySumMS: 70} + first.LatencyCounts[1], first.LatencyCounts[2] = 1, 1 + second := observability.RequestBucket{Start: start, RouteFamily: "sessions", Method: "GET", Outcome: "success", Count: 1, LatencySumMS: 300} + second.LatencyCounts[5] = 1 + if err := s.WriteRequestBuckets(t.Context(), []observability.RequestBucket{first}, 0, 0); err != nil { + t.Fatal(err) + } + if err := s.WriteRequestBuckets(t.Context(), []observability.RequestBucket{second}, 1, 0); err != nil { + t.Fatal(err) + } + got, err := s.ReadOperatorMetrics(t.Context(), start, start.Add(time.Minute), time.Minute) + if err != nil { + t.Fatal(err) + } + if len(got.Requests) != 1 || got.Requests[0].Count != 3 || got.Requests[0].LatencySumMS != 370 || + got.Requests[0].LatencyCounts[1] != 1 || got.Requests[0].LatencyCounts[2] != 1 || got.Requests[0].LatencyCounts[5] != 1 { + t.Fatalf("request bucket lost concurrent-shape additions: %+v", got.Requests) + } + // Collector health is written in the current minute, separate from the + // completed request's event-time bucket. + health, err := s.ReadOperatorMetrics(t.Context(), start, start.Add(2*time.Minute), time.Minute) + if err != nil || len(health.Collector) == 0 { + t.Fatalf("collector health unavailable: %+v, %v", health.Collector, err) + } +} diff --git a/services/agents-api/internal/store/observability_retention.go b/services/agents-api/internal/store/observability_retention.go new file mode 100644 index 000000000..ff65fa4ea --- /dev/null +++ b/services/agents-api/internal/store/observability_retention.go @@ -0,0 +1,29 @@ +package store + +import ( + "context" + "time" +) + +// PruneOperatorMetrics deletes small batches so retention never holds a long lock. +func (s *Store) PruneOperatorMetrics(ctx context.Context, now time.Time) error { + cutoffs := []struct { + query string + before time.Time + }{ + {`DELETE FROM observability_request_minute_buckets WHERE ctid IN + (SELECT ctid FROM observability_request_minute_buckets WHERE bucket_start < $1 LIMIT 256 FOR UPDATE SKIP LOCKED)`, now.Add(-30 * 24 * time.Hour)}, + {`DELETE FROM observability_collector_minute_buckets WHERE ctid IN + (SELECT ctid FROM observability_collector_minute_buckets WHERE bucket_start < $1 LIMIT 256 FOR UPDATE SKIP LOCKED)`, now.Add(-30 * 24 * time.Hour)}, + {`DELETE FROM observability_model_attempts WHERE id IN + (SELECT id FROM observability_model_attempts WHERE started_at < $1 LIMIT 256 FOR UPDATE SKIP LOCKED)`, now.Add(-7 * 24 * time.Hour)}, + {`DELETE FROM observability_tool_attempts WHERE id IN + (SELECT id FROM observability_tool_attempts WHERE finished_at < $1 LIMIT 256 FOR UPDATE SKIP LOCKED)`, now.Add(-7 * 24 * time.Hour)}, + } + for _, cutoff := range cutoffs { + if _, err := s.pool.Exec(ctx, cutoff.query, cutoff.before); err != nil { + return err + } + } + return nil +} diff --git a/services/agents-api/internal/store/observability_tools.go b/services/agents-api/internal/store/observability_tools.go new file mode 100644 index 000000000..f6ef4f728 --- /dev/null +++ b/services/agents-api/internal/store/observability_tools.go @@ -0,0 +1,49 @@ +package store + +import ( + "context" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" +) + +const insertToolAttempt = `INSERT INTO observability_tool_attempts + (id, tenant_id, session_id, turn_id, started_at, finished_at, tool_category, outcome, duration_ms) +SELECT $1::uuid, s.tenant_id, s.id, t.id, $5::timestamptz, $6, $7, $8, $9::bigint +FROM sessions s JOIN turns t ON t.session_id = s.id +WHERE s.tenant_id = $2::uuid AND s.id = $3::uuid AND t.id = $4::uuid +ON CONFLICT (id) DO NOTHING` + +// WriteToolAttempts stores only sanitized terminal tool evidence. Duplicate +// native completion events have one deterministic attempt ID. +func (s *Store) WriteToolAttempts(ctx context.Context, attempts []observability.ToolAttempt, dropped, failed int64) error { + tx, err := s.pool.Begin(ctx) + if err != nil { + return err + } + defer tx.Rollback(ctx) + var inserted int64 + for _, attempt := range attempts { + result, err := tx.Exec(ctx, insertToolAttempt, attempt.ID, attempt.TenantID, attempt.SessionID, attempt.TurnID, + attempt.StartedAt, attempt.FinishedAt, attempt.Category, attempt.Outcome, attempt.DurationMS) + if err != nil { + return err + } + inserted += result.RowsAffected() + } + if len(attempts) > 0 || dropped > 0 || failed > 0 { + _, err = tx.Exec(ctx, `INSERT INTO observability_collector_minute_buckets AS current + (bucket_start, source, attempted_count, observed_count, dropped_count, export_failed_count) + VALUES ($1, 'tool', $2, $3, $4, $5) + ON CONFLICT (bucket_start, source) DO UPDATE SET + attempted_count = current.attempted_count + EXCLUDED.attempted_count, + observed_count = current.observed_count + EXCLUDED.observed_count, + dropped_count = current.dropped_count + EXCLUDED.dropped_count, + export_failed_count = current.export_failed_count + EXCLUDED.export_failed_count`, + time.Now().UTC().Truncate(time.Minute), int64(len(attempts))+dropped, inserted, dropped, failed) + if err != nil { + return err + } + } + return tx.Commit(ctx) +} diff --git a/services/agents-api/internal/store/observability_tools_test.go b/services/agents-api/internal/store/observability_tools_test.go new file mode 100644 index 000000000..a98c61a70 --- /dev/null +++ b/services/agents-api/internal/store/observability_tools_test.go @@ -0,0 +1,55 @@ +package store + +import ( + "testing" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" + "github.com/google/uuid" +) + +func TestPostgresToolAttemptIsIdempotentAndTenantScoped(t *testing.T) { + s, pool := testStore(t) + tenant, session, turn, attemptID := uuid.NewString(), uuid.NewString(), uuid.NewString(), uuid.NewString() + _, err := pool.Exec(t.Context(), `INSERT INTO sessions (id, tenant_id, engine, idempotency_key, request_hash) + VALUES ($1, $2, 'codex', $3, 'observability-test')`, session, tenant, uuid.NewString()) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { + _, _ = pool.Exec(t.Context(), `DELETE FROM turns WHERE id=$1`, turn) + _, _ = pool.Exec(t.Context(), `DELETE FROM sessions WHERE id=$1`, session) + }) + _, err = pool.Exec(t.Context(), `INSERT INTO turns (id, session_id, status, completed_at) + VALUES ($1, $2, 'completed', clock_timestamp())`, turn, session) + if err != nil { + t.Fatal(err) + } + finished := time.Now().UTC() + duration := int64(120) + value := observability.ToolAttempt{ID: attemptID, TenantID: tenant, SessionID: session, TurnID: turn, + FinishedAt: finished, Category: "command", Outcome: "success", DurationMS: &duration} + if err := s.WriteToolAttempts(t.Context(), []observability.ToolAttempt{value}, 0, 0); err != nil { + t.Fatal(err) + } + if err := s.WriteToolAttempts(t.Context(), []observability.ToolAttempt{value}, 0, 0); err != nil { + t.Fatal(err) + } + var count int + var startedAt *time.Time + if err := pool.QueryRow(t.Context(), `SELECT count(*), max(started_at) FROM observability_tool_attempts WHERE id=$1`, attemptID).Scan(&count, &startedAt); err != nil || count != 1 || startedAt != nil { + t.Fatalf("tool attempt deduplication or unknown start failed: count=%d start=%v err=%v", count, startedAt, err) + } + value.ID = uuid.NewString() + value.TenantID = uuid.NewString() + if err := s.WriteToolAttempts(t.Context(), []observability.ToolAttempt{value}, 0, 0); err != nil { + t.Fatal(err) + } + if err := pool.QueryRow(t.Context(), `SELECT count(*) FROM observability_tool_attempts WHERE id=$1`, value.ID).Scan(&count); err != nil || count != 0 { + t.Fatalf("foreign tenant wrote an attempt: count=%d err=%v", count, err) + } + metrics, err := s.ReadOperatorMetrics(t.Context(), finished.Add(-time.Minute), finished.Add(time.Minute), time.Minute) + if err != nil || len(metrics.Tools) == 0 || len(metrics.Turns) == 0 { + t.Fatalf("terminal operator metrics unavailable: tools=%+v turns=%+v err=%v", metrics.Tools, metrics.Turns, err) + } +} diff --git a/services/agents-api/internal/store/runtime_history_acceptance_test.go b/services/agents-api/internal/store/runtime_history_acceptance_test.go index 23cc3aa7f..2c45688cb 100644 --- a/services/agents-api/internal/store/runtime_history_acceptance_test.go +++ b/services/agents-api/internal/store/runtime_history_acceptance_test.go @@ -148,10 +148,10 @@ func TestPostgresRuntimeHistoryDenseReadAndBoundedRetention(t *testing.T) { } // Seed the equivalent of a full day of five-second periodic samples in one // fixture statement; production inserts still go through the exporter. - _, err := pool.Exec(t.Context(), `INSERT INTO runtime_history_samples + _, err := pool.Exec(t.Context(), `INSERT INTO observability_runtime_samples (tenant_id,session_id,environment_id,resolved_at_ns,allocation_id,provider_type,status,observed_at_ns,started_at_ns,cpu_usage_seconds,cpu_capacity_cores,memory_usage_bytes,input_tokens,output_tokens) SELECT tenant_id,session_id,environment_id,resolved_at_ns+n*5000000000,allocation_id,provider_type,status,observed_at_ns+n*5000000000,started_at_ns,n::double precision,cpu_capacity_cores,memory_usage_bytes,input_tokens+n,output_tokens+n - FROM runtime_history_samples CROSS JOIN generate_series(1,17279) n WHERE tenant_id=$1 AND session_id=$2`, scope.TenantID, scope.SessionID) + FROM observability_runtime_samples CROSS JOIN generate_series(1,17279) n WHERE tenant_id=$1 AND session_id=$2`, scope.TenantID, scope.SessionID) if err != nil { t.Fatal(err) } @@ -173,10 +173,10 @@ func TestPostgresRuntimeHistoryDenseReadAndBoundedRetention(t *testing.T) { if err := s.InsertRuntimeHistorySample(t.Context(), old); err != nil { t.Fatal(err) } - _, err = pool.Exec(t.Context(), `INSERT INTO runtime_history_samples + _, err = pool.Exec(t.Context(), `INSERT INTO observability_runtime_samples (tenant_id,session_id,environment_id,resolved_at_ns,allocation_id,provider_type,status,observed_at_ns,started_at_ns) SELECT tenant_id,session_id,environment_id,resolved_at_ns-n,allocation_id,provider_type,'unavailable',NULL,NULL - FROM runtime_history_samples CROSS JOIN generate_series(1,4100) n WHERE tenant_id=$1 AND session_id=$2 AND resolved_at_ns=$3`, scope.TenantID, scope.SessionID, old.ResolvedAt.UnixNano()) + FROM observability_runtime_samples CROSS JOIN generate_series(1,4100) n WHERE tenant_id=$1 AND session_id=$2 AND resolved_at_ns=$3`, scope.TenantID, scope.SessionID, old.ResolvedAt.UnixNano()) if err != nil { t.Fatal(err) } @@ -188,13 +188,13 @@ func TestPostgresRuntimeHistoryDenseReadAndBoundedRetention(t *testing.T) { t.Fatal(err) } var remaining int - if err := pool.QueryRow(t.Context(), `SELECT count(*) FROM runtime_history_samples WHERE tenant_id=$1 AND resolved_at_ns<$2`, scope.TenantID, end.Add(-7*24*time.Hour).UnixNano()).Scan(&remaining); err != nil || remaining != 5 { + if err := pool.QueryRow(t.Context(), `SELECT count(*) FROM observability_runtime_samples WHERE tenant_id=$1 AND resolved_at_ns<$2`, scope.TenantID, end.Add(-7*24*time.Hour).UnixNano()).Scan(&remaining); err != nil || remaining != 5 { t.Fatal("cleanup exceeded bounded batch", remaining, err) } if err := reader.Prune(t.Context()); err != nil { t.Fatal(err) } - if err := pool.QueryRow(t.Context(), `SELECT count(*) FROM runtime_history_samples WHERE tenant_id=$1 AND resolved_at_ns<$2`, scope.TenantID, end.Add(-7*24*time.Hour).UnixNano()).Scan(&remaining); err != nil || remaining != 0 { + if err := pool.QueryRow(t.Context(), `SELECT count(*) FROM observability_runtime_samples WHERE tenant_id=$1 AND resolved_at_ns<$2`, scope.TenantID, end.Add(-7*24*time.Hour).UnixNano()).Scan(&remaining); err != nil || remaining != 0 { t.Fatal("retention cleanup incomplete", remaining, err) } cancelled, cancel := context.WithCancel(t.Context()) diff --git a/services/agents-api/migrations/000066_observability.sql b/services/agents-api/migrations/000066_observability.sql new file mode 100644 index 000000000..263db79d3 --- /dev/null +++ b/services/agents-api/migrations/000066_observability.sql @@ -0,0 +1,78 @@ +-- +goose Up +-- Runtime samples remain a bounded telemetry projection, not execution or Usage authority. +ALTER TABLE runtime_history_samples RENAME TO observability_runtime_samples; +ALTER INDEX runtime_history_samples_pkey RENAME TO observability_runtime_samples_pkey; +ALTER INDEX runtime_history_samples_retention_idx RENAME TO observability_runtime_samples_retention_idx; +CREATE INDEX turns_completed_time_idx ON turns (completed_at) WHERE completed_at IS NOT NULL; + +-- Histogram bins: <=10, 25, 50, 100, 250, 500, 1000, 2500, 5000, +-- 10000 milliseconds, and >10000 milliseconds. Counts are disjoint. +CREATE TABLE observability_request_minute_buckets ( + bucket_start timestamptz NOT NULL, + route_family text NOT NULL CHECK (length(route_family) BETWEEN 1 AND 64), + method text NOT NULL CHECK (method IN ('GET', 'POST', 'PUT', 'PATCH', 'DELETE', 'HEAD', 'OTHER')), + outcome text NOT NULL CHECK (outcome IN ('success', 'client_error', 'server_error')), + request_count bigint NOT NULL DEFAULT 0 CHECK (request_count >= 0), + latency_sum_ms bigint NOT NULL DEFAULT 0 CHECK (latency_sum_ms >= 0), + latency_bucket_counts bigint[] NOT NULL DEFAULT array_fill(0::bigint, ARRAY[11]) + CHECK (array_length(latency_bucket_counts, 1) = 11), + PRIMARY KEY (bucket_start, route_family, method, outcome) +); + +CREATE TABLE observability_model_attempts ( + id uuid PRIMARY KEY, + tenant_id uuid NOT NULL, + session_id uuid NOT NULL REFERENCES sessions(id) ON DELETE CASCADE, + turn_id uuid NOT NULL REFERENCES turns(id) ON DELETE CASCADE, + started_at timestamptz NOT NULL, + finished_at timestamptz NOT NULL CHECK (finished_at >= started_at), + model_family text NOT NULL CHECK (length(model_family) BETWEEN 1 AND 64), + provider_type text NOT NULL CHECK (length(provider_type) BETWEEN 1 AND 64), + outcome text NOT NULL CHECK (outcome IN ('success', 'error', 'cancelled', 'unknown')), + duration_ms bigint NOT NULL CHECK (duration_ms >= 0), + input_tokens bigint CHECK (input_tokens >= 0), + output_tokens bigint CHECK (output_tokens >= 0), + usage_status text NOT NULL CHECK (usage_status IN ('observed', 'unavailable', 'unsupported')), + CHECK ((input_tokens IS NULL AND output_tokens IS NULL) OR usage_status = 'observed') +); +CREATE INDEX observability_model_attempts_tenant_time_idx ON observability_model_attempts (tenant_id, started_at DESC); +CREATE INDEX observability_model_attempts_session_time_idx ON observability_model_attempts (session_id, started_at DESC); +CREATE INDEX observability_model_attempts_retention_idx ON observability_model_attempts (started_at); + +CREATE TABLE observability_tool_attempts ( + id uuid PRIMARY KEY, + tenant_id uuid NOT NULL, + session_id uuid NOT NULL REFERENCES sessions(id) ON DELETE CASCADE, + turn_id uuid NOT NULL REFERENCES turns(id) ON DELETE CASCADE, + started_at timestamptz, + finished_at timestamptz NOT NULL, + tool_category text NOT NULL CHECK (length(tool_category) BETWEEN 1 AND 64), + outcome text NOT NULL CHECK (outcome IN ('success', 'error', 'cancelled', 'unknown')), + duration_ms bigint CHECK (duration_ms >= 0), + CHECK (started_at IS NULL OR finished_at >= started_at) +); +CREATE INDEX observability_tool_attempts_tenant_time_idx ON observability_tool_attempts (tenant_id, finished_at DESC); +CREATE INDEX observability_tool_attempts_session_time_idx ON observability_tool_attempts (session_id, finished_at DESC); +CREATE INDEX observability_tool_attempts_retention_idx ON observability_tool_attempts (finished_at); + +CREATE TABLE observability_collector_minute_buckets ( + bucket_start timestamptz NOT NULL, + source text NOT NULL CHECK (source IN ('runtime', 'request', 'model', 'tool', 'export')), + attempted_count bigint NOT NULL DEFAULT 0 CHECK (attempted_count >= 0), + observed_count bigint NOT NULL DEFAULT 0 CHECK (observed_count >= 0), + unavailable_count bigint NOT NULL DEFAULT 0 CHECK (unavailable_count >= 0), + timeout_count bigint NOT NULL DEFAULT 0 CHECK (timeout_count >= 0), + dropped_count bigint NOT NULL DEFAULT 0 CHECK (dropped_count >= 0), + export_failed_count bigint NOT NULL DEFAULT 0 CHECK (export_failed_count >= 0), + PRIMARY KEY (bucket_start, source) +); + +-- +goose Down +DROP TABLE observability_collector_minute_buckets; +DROP TABLE observability_tool_attempts; +DROP TABLE observability_model_attempts; +DROP TABLE observability_request_minute_buckets; +DROP INDEX turns_completed_time_idx; +ALTER INDEX observability_runtime_samples_retention_idx RENAME TO runtime_history_samples_retention_idx; +ALTER INDEX observability_runtime_samples_pkey RENAME TO runtime_history_samples_pkey; +ALTER TABLE observability_runtime_samples RENAME TO runtime_history_samples; diff --git a/services/agents-api/tests/runtime_history_live.py b/services/agents-api/tests/runtime_history_live.py index ba06f92fc..3e0c44017 100644 --- a/services/agents-api/tests/runtime_history_live.py +++ b/services/agents-api/tests/runtime_history_live.py @@ -33,7 +33,7 @@ def request(self, *args, **kwargs): encoded = json.dumps(value) base.require(not any(secret in encoded for secret in self.secrets), "public_response_exposed_secret") base.require(not any(marker in encoded for marker in ( - "postgres://", "postgresql://", "runtime_history_samples", "token_sha256", + "postgres://", "postgresql://", "observability_runtime_samples", "token_sha256", "unix:///", "host.microsandbox.internal", "/home/parsar-acceptance/")), "public_response_exposed_backend_configuration") return value diff --git a/services/core-console/sandbox_admin.go b/services/core-console/sandbox_admin.go index cfcdd48fc..6daa3e05a 100644 --- a/services/core-console/sandbox_admin.go +++ b/services/core-console/sandbox_admin.go @@ -39,6 +39,11 @@ func sandboxAdminRequest(r *http.Request) bool { return false } +// Operator metrics expose only the fixed, read-only deployment summary. +func observabilityAdminRequest(r *http.Request) bool { + return r.URL.Path == "/core/v1/observability/summary" && r.Method == http.MethodGet +} + func explicitBearer(r *http.Request) bool { parts := strings.Fields(r.Header.Get("Authorization")) return len(r.Header.Values("Authorization")) == 1 && len(parts) == 2 && strings.EqualFold(parts[0], "Bearer") diff --git a/services/core-console/sandbox_admin_test.go b/services/core-console/sandbox_admin_test.go index 886427fd2..14f6c5e1c 100644 --- a/services/core-console/sandbox_admin_test.go +++ b/services/core-console/sandbox_admin_test.go @@ -44,6 +44,7 @@ func TestSandboxAdminProxyPreservesIndependentCredential(t *testing.T) { {"DELETE", "/core/v1/sandbox/nodes/node-id"}, {"GET", "/core/v1/sandbox/nodes/node-id/allocations"}, {"POST", "/core/v1/sandbox/enrollment-tokens"}, + {"GET", "/core/v1/observability/summary?range=1h"}, } { t.Run(tc.method+tc.path, func(t *testing.T) { r := adminRequest(t, server.URL, tc.method, tc.path, "separate-admin-key") @@ -56,7 +57,7 @@ func TestSandboxAdminProxyPreservesIndependentCredential(t *testing.T) { } }) } - if calls.Load() != 6 { + if calls.Load() != 7 { t.Fatalf("proxied %d operations", calls.Load()) } } @@ -129,6 +130,8 @@ func TestSandboxNodeControlAndUnknownCoreRoutesAreNotProxied(t *testing.T) { {"POST", "/core/v1/sandbox/nodes"}, {"GET", "/core/v1/sandbox/enrollment-tokens"}, {"GET", "/core/v1/sandbox/nodes/id/allocations/extra"}, + {"POST", "/core/v1/observability/summary"}, + {"GET", "/core/v1/observability/summary/extra"}, } { t.Run(tc.method+tc.path, func(t *testing.T) { r := adminRequest(t, server.URL, tc.method, tc.path, "separate-admin-key") diff --git a/services/core-console/server.go b/services/core-console/server.go index 293c1193a..9b0018d34 100644 --- a/services/core-console/server.go +++ b/services/core-console/server.go @@ -72,9 +72,9 @@ func newConsole(c config) (*console, error) { r.Out.Header.Del("Cookie") r.Out.Header.Del("Origin") r.Out.Header.Del("Referer") - if ((publicAPIRequest(r.In) || projectExtensionRequest(r.In)) && explicitBearer(r.In)) || nodeTransportRequest(r.In) || (sandboxAdminRequest(r.In) && c.adminToken == "") { + if ((publicAPIRequest(r.In) || projectExtensionRequest(r.In)) && explicitBearer(r.In)) || nodeTransportRequest(r.In) || ((sandboxAdminRequest(r.In) || observabilityAdminRequest(r.In)) && c.adminToken == "") { r.Out.Header.Set("Authorization", r.In.Header.Get("Authorization")) - } else if sandboxAdminRequest(r.In) || projectKeyAdminRequest(r.In) { + } else if sandboxAdminRequest(r.In) || observabilityAdminRequest(r.In) || projectKeyAdminRequest(r.In) { r.Out.Header.Set("Authorization", "Bearer "+c.adminToken) } else { r.Out.Header.Set("Authorization", "Bearer "+c.token) @@ -170,7 +170,7 @@ func (h *console) ServeHTTP(w http.ResponseWriter, r *http.Request) { return } if (r.URL.Path == "/core" || strings.HasPrefix(r.URL.Path, "/core/")) && h.adminToken == "" && !projectExtensionRequest(r) { - if !sandboxAdminRequest(r) { + if !sandboxAdminRequest(r) && !observabilityAdminRequest(r) { http.NotFound(w, r) return } @@ -206,7 +206,7 @@ func (h *console) ServeHTTP(w http.ResponseWriter, r *http.Request) { return } if r.URL.Path == "/core" || strings.HasPrefix(r.URL.Path, "/core/") { - if !sandboxAdminRequest(r) && !projectExtensionRequest(r) { + if !sandboxAdminRequest(r) && !observabilityAdminRequest(r) && !projectExtensionRequest(r) { http.NotFound(w, r) return }