diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 91fcfabc5..6cc5fface 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -163,6 +163,17 @@ logs. Update this guide when architecture, ownership or generated contracts chan Comments and documentation are English. Reuse existing helpers and error mapping; split oversized components before extending them. Use `internal/obs/log` for logs. +Operator observability is a side-channel owned by Core. The authenticated +`/core/v1/admin/observability` reads bounded request, terminal Turn, +tool attempt, and collector aggregates; the Core Web uses it through +`packages/agents-client`'s `AdminClient`, alongside the existing Core service +metrics endpoint. Runtime samples retain their separate public history +contract under the `observability_runtime_samples` table. Native model attempts +have a reserved table but no qualified producer; configured models, cumulative +Usage, and Turns must not be counted as model calls. Missing collector coverage +stays unavailable rather than becoming zero. See +`contracts/agents-api/system-observability-plan.md` for source and retention details. + ## Required checks Run `make check` before completion. The standalone gate includes all daemon/shared diff --git a/apps/web/e2e/agents-lifecycle.spec.ts b/apps/web/e2e/agents-lifecycle.spec.ts index 9de471d89..4f829fbb1 100644 --- a/apps/web/e2e/agents-lifecycle.spec.ts +++ b/apps/web/e2e/agents-lifecycle.spec.ts @@ -210,6 +210,48 @@ test("opens Dashboard as the default landing page", async ({ page, request }) => await expect(page.locator(".dashboard-page")).toBeVisible(); }); +test("shows operator tables and keeps incomplete request coverage unavailable", async ({ page, request }, testInfo) => { + await resetFixture(request); + await page.route("**/console/config", (route) => route.fulfill({ json: { + sandbox_admin: true, node_installer: false, node_installer_sha256: "", + } })); + const end = new Date(Math.floor(Date.now() / 60_000) * 60_000); + const start = new Date(end.getTime() - 60 * 60_000); + await page.route("**/core/v1/admin/observability?range=1h", (route) => route.fulfill({ json: { + generated_at: end.toISOString(), start: start.toISOString(), end: end.toISOString(), step_seconds: 60, + requests: [{ start: start.toISOString(), route_family: "sessions", outcome: "server_error", count: 2, + latency_sum_ms: 120, latency_bucket_counts: [0, 0, 0, 2, 0, 0, 0, 0, 0, 0, 0] }], + collector: [{ start: start.toISOString(), source: "request", attempted_count: 2, observed_count: 2, + unavailable_count: 0, timeout_count: 0, dropped_count: 0, export_failed_count: 0 }], + turns: [{ start: start.toISOString(), status: "failed", count: 1, queue_p95_ms: 50, execution_p95_ms: 400 }], + tools: [{ start: start.toISOString(), category: "command", outcome: "error", count: 1, + timed_count: 1, duration_p95_ms: 300 }], + } })); + await page.route("**/core/v1/admin/core-metrics?range=1h", (route) => route.fulfill({ json: { + object: "core.metrics", range: { start: start.toISOString(), end: end.toISOString(), resolution_seconds: 60 }, + service: { status: "healthy", revision: null, started_at: start.toISOString(), execution_owner: true }, + execution: { slots_in_use: 1, slots_total: 4, queued_turns: 3, waiting_for_daemon: 1, + in_progress_turns: 2, oldest_queued_seconds: 12, connected_daemons: 2, interrupted: 0, + unavailable: 0, queue_wait_ms: { p50: 20, p95: 50 }, series: [] }, + database: { ping_ms: { p50: 3, p95: 7 }, pool: { in_use: 2, idle: 4, max: 20 }, size_bytes: 1024, series: [] }, + jobs: [{ id: "scheduler", status: "running", last_run_at: end.toISOString(), processed: 5, failed: 0 }], + process: { memory_bytes: 1048576, goroutines: 42 }, + } })); + await page.goto("/"); + await page.getByRole("tab", { name: "Observability" }).click(); + const system = page.getByRole("region", { name: "System observability" }); + await expect(system.getByRole("heading", { name: "System observability" })).toBeVisible(); + await expect(system.getByRole("heading", { name: "Terminal Turns" })).toBeVisible(); + await expect(system.getByRole("heading", { name: "Actual model invocations" })).toBeVisible(); + await expect(system.getByRole("heading", { name: "Sandbox nodes" })).toBeVisible(); + await expect(page.getByRole("heading", { name: "Core service" })).toBeVisible(); + await expect(page.getByRole("heading", { name: "Background jobs" })).toBeVisible(); + await expect(system.getByText("Unavailable", { exact: true }).first()).toBeVisible(); + await expect(system.getByRole("rowheader", { name: "sessions" })).toBeVisible(); + await expect(system.getByRole("rowheader", { name: "command" })).toBeVisible(); + await attachElementScreenshot(page.locator(".dashboard-page"), testInfo, "system-observability-dashboard"); +}); + test("explains a 502 Core backend failure and opens copyable Docker recovery steps", async ({ page }) => { await page.route("**/v1/agents**", async (route) => { await route.fulfill({ @@ -2388,9 +2430,8 @@ test("presents Dashboard page-chain results and System boundaries without extra await expect(dashboard).toContainText("Latest complete paginated reads"); await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "Agents" })).toContainText("3"); await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "Sessions" })).toContainText("1"); - await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "In progress" })).toContainText("0"); - await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "Needs attention" })).toContainText("0"); - await expect(dashboard).not.toContainText("Reported aggregate tokens"); + await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "Active sandboxes" })).toBeVisible(); + await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "Reported tokens" })).toBeVisible(); await expect(dashboard).toContainText("No Sessions currently need attention."); await expect(dashboard.getByRole("button", { name: /Create agent/ })).toBeVisible(); await expect(dashboard.getByRole("button", { name: /Start session/ })).toBeVisible(); @@ -2683,7 +2724,13 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand await page.goto("/"); const dashboard = page.locator(".dashboard-page"); await expect(dashboard.getByRole("heading", { name: "Dashboard", exact: true })).toBeVisible(); + await dashboard.getByRole("tab", { name: "Observability" }).click(); const refresh = dashboard.getByRole("button", { name: "Refresh Dashboard snapshot" }); + const sandboxDiagnostics = dashboard.getByRole("region", { name: "Sandbox diagnostics" }); + await expect(sandboxDiagnostics).toBeVisible(); + await expect(sandboxDiagnostics).toContainText("12 managed targets"); + await expect(sandboxDiagnostics).toContainText("12 with measured usage and limit"); + await expect(sandboxDiagnostics).toContainText("Highest memory pressure"); await expect(dashboard.locator(".dashboard-runtime-sample-count")).toContainText("1 sample ·"); await expect(dashboard.getByRole("heading", { name: "CPU usage" })).toBeVisible(); await expect(dashboard.getByRole("heading", { name: "Memory usage" })).toBeVisible(); @@ -2735,7 +2782,7 @@ test("renders Runtime telemetry as visual snapshot panels with details on demand await expect(cpuCard.locator(".dashboard-runtime-trend-tooltip")).toBeVisible(); await cpuChart.click({ position: { x: 260, y: 90 } }); await expect(cpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("Pinned"); - await expect(cpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("10%"); + await expect(cpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("Unavailable"); await cpuChart.focus(); await cpuChart.press("ArrowRight"); await expect(cpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("Pinned"); @@ -2913,6 +2960,7 @@ test("restores retained Runtime history after a Dashboard reload", async ({ page }); await page.goto("/"); + await page.getByRole("tab", { name: "Observability" }).click(); const dashboard = page.locator(".dashboard-runtime-panel"); await expect(dashboard.getByRole("group", { name: "Runtime trend source" })).toHaveCount(0); await expect(dashboard.getByLabel(/Durable · 30s; 1 Runtime targets/)).toBeVisible(); @@ -2930,9 +2978,9 @@ test("restores retained Runtime history after a Dashboard reload", async ({ page const durableCpuCard = durableCpuChart.locator("xpath=ancestor::section[contains(@class, 'dashboard-runtime-trend-card')]"); await durableCpuChart.focus(); await durableCpuChart.press("ArrowLeft"); - await expect(durableCpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("0%"); + await expect(durableCpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("Unavailable"); await durableCpuChart.press("ArrowLeft"); - await expect(durableCpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("0%"); + await expect(durableCpuCard.locator(".dashboard-runtime-trend-tooltip")).toContainText("Unavailable"); const durableMemoryCard = dashboard.getByRole("region", { name: "Memory usage durable history chart" }); const durableMemorySpan = await durableMemoryCard.locator("canvas").evaluate((canvas: HTMLCanvasElement) => { const context = canvas.getContext("2d"); @@ -2973,6 +3021,7 @@ test("restores retained Runtime history after a Dashboard reload", async ({ page await expect(refreshedDurableCpuChart).toHaveAttribute("data-view-end", String(durableZoomEnd)); await page.reload(); + await page.getByRole("tab", { name: "Observability" }).click(); await expect(dashboard.getByRole("group", { name: "Runtime trend source" })).toHaveCount(0); await expect(dashboard.getByLabel(/Durable · 30s; 1 Runtime targets/)).toBeVisible(); await expect(dashboard.getByText("CPU usage durable trend available")).toBeAttached(); diff --git a/apps/web/e2e/sandbox-setup.spec.ts b/apps/web/e2e/sandbox-setup.spec.ts index c58cd027d..c5fd35b88 100644 --- a/apps/web/e2e/sandbox-setup.spec.ts +++ b/apps/web/e2e/sandbox-setup.spec.ts @@ -189,13 +189,15 @@ for (const operation of ["setup", "enrollment"] as const) { } test("unpaired or unavailable consoles show setup guidance without admin credentials or manager requests", async ({ page, request }) => { + await expect.poll(async () => ((await (await request.get(`${fixture}/__fixture/sandbox`)).json()) as { calls: unknown[] }).calls.length).toBeGreaterThan(0); + const initialCalls = ((await (await request.get(`${fixture}/__fixture/sandbox`)).json()) as { calls: unknown[] }).calls.length; for (const body of [{ sandbox_admin: false, node_installer: true }, null]) { await page.route("**/console/config", (route) => route.fulfill(body ? { contentType: "application/json", body: JSON.stringify(body) } : { status: 404, body: "Not found" })); await page.reload(); await page.getByRole("button", { name: "Hosted Sandbox Manager", exact: true }).click(); await expect(page.getByRole("alert")).toContainText("Sandbox administration is not configured"); await expect(page.getByLabel("Deployment admin key")).toHaveCount(0); - expect((await (await request.get(`${fixture}/__fixture/sandbox`)).json()).calls).toHaveLength(0); + expect((await (await request.get(`${fixture}/__fixture/sandbox`)).json()).calls).toHaveLength(initialCalls); } await page.route("**/console/config", (route) => route.fulfill({ contentType: "application/json", body: JSON.stringify({ sandbox_admin: true, node_installer: true, node_installer_sha256: "a".repeat(64) }) })); await page.getByRole("button", { name: "Refresh sandbox state" }).click(); diff --git a/apps/web/src/App.tsx b/apps/web/src/App.tsx index 7765467be..8464c582d 100644 --- a/apps/web/src/App.tsx +++ b/apps/web/src/App.tsx @@ -2367,6 +2367,7 @@ export function App() { runtimeCollectionError={runtimeCollectionError} runtimeCollectionHasSnapshot={runtimeCollectionHasSnapshot} loadRuntimeHistory={loadDashboardRuntimeHistory} + coreBaseUrl={connection.baseUrl} onRefresh={refreshDashboard} onCreateAgent={openAgentSetup} onStartSession={() => openSessionSetup()} diff --git a/apps/web/src/features/dashboard/CoreServiceMetricsContent.tsx b/apps/web/src/features/dashboard/CoreServiceMetricsContent.tsx new file mode 100644 index 000000000..8b09c3e17 --- /dev/null +++ b/apps/web/src/features/dashboard/CoreServiceMetricsContent.tsx @@ -0,0 +1,70 @@ +import { useEffect, useState } from "react"; +import { useTranslation } from "react-i18next"; +import { AdminClient, type CoreMetricsView, type OperatorMetricsRange } from "@agents-core-web/agents-client"; + +import { isLocalProxyBaseUrl } from "../../lib/connection"; +import { formatDashboardBytes } from "./dashboard-model"; + +export function CoreServiceMetricsContent({ coreBaseUrl, range, refresh }: { coreBaseUrl: string; range: OperatorMetricsRange; refresh: number }) { + const { t, i18n } = useTranslation("dashboard"); + const localSource = isLocalProxyBaseUrl(coreBaseUrl); + const [metrics, setMetrics] = useState(null); + const [failed, setFailed] = useState(false); + useEffect(() => { + const controller = new AbortController(); + if (!localSource) { + setMetrics(null); + setFailed(false); + return () => controller.abort(); + } + setMetrics(null); + setFailed(false); + void new AdminClient().retrieveCoreMetrics(range, { signal: controller.signal }).then((value) => { + if (!controller.signal.aborted) { setMetrics(value); setFailed(false); } + }).catch(() => { + if (!controller.signal.aborted) { setMetrics(null); setFailed(true); } + }); + return () => controller.abort(); + }, [localSource, range, refresh]); + const unavailable = t("system.unavailableValue"); + const number = (value: number | null | undefined) => value == null ? unavailable : value.toLocaleString(i18n.resolvedLanguage); + const milliseconds = (value: number | null | undefined) => value == null ? unavailable : `${number(Math.round(value))} ms`; + const execution = metrics?.execution; + const database = metrics?.database; + const process = metrics?.process; + const rows: Array<[string, string]> = [ + [t("core.queued"), number(execution?.queued_turns)], + [t("core.running"), number(execution?.in_progress_turns)], + [t("core.waitingDaemon"), number(execution?.waiting_for_daemon)], + [t("core.connectedDaemons"), number(execution?.connected_daemons)], + [t("core.oldestQueue"), execution?.oldest_queued_seconds == null ? unavailable : `${number(Math.round(execution.oldest_queued_seconds))} s`], + [t("core.queueP95"), milliseconds(execution?.queue_wait_ms.p95)], + [t("core.interrupted"), number(execution?.interrupted)], + [t("core.unavailable"), number(execution?.unavailable)], + ]; + return
+
+

{t("core.title")}

+

{t("core.subtitle")}

+
{metrics ? t("core.status", { status: metrics.service.status }) : !localSource ? t("system.adminUnavailable") : failed ? t("system.stale") : t("system.loading")}
+
+
{t("core.slots")}{execution?.slots_in_use == null || execution.slots_total == null ? unavailable : `${number(execution.slots_in_use)}/${number(execution.slots_total)}`}{t("core.executionOwner", { status: metrics?.service.execution_owner == null ? unavailable : metrics.service.execution_owner ? t("core.yes") : t("core.no") })}
+
{t("core.queue")}{number(execution?.queued_turns)}{t("core.runningCount", { value: number(execution?.in_progress_turns) })}
+
{t("core.databasePing")}{milliseconds(database?.ping_ms.p95)}{t("core.pool", { used: number(database?.pool.in_use), max: number(database?.pool.max) })}
+
{t("core.processMemory")}{process?.memory_bytes == null ? unavailable : formatDashboardBytes(process.memory_bytes)}{t("core.goroutines", { value: number(process?.goroutines) })}
+
+
+

{t("core.executionTable")}

{rows.map(([label, value]) => )}
{label}{value}
+

{t("core.databaseTable")}

+ + + + + +
{t("core.pingP50")}{milliseconds(database?.ping_ms.p50)}
{t("core.pingP95")}{milliseconds(database?.ping_ms.p95)}
{t("core.poolInUse")}{number(database?.pool.in_use)}
{t("core.poolIdle")}{number(database?.pool.idle)}
{t("core.databaseSize")}{database?.size_bytes == null ? unavailable : formatDashboardBytes(database.size_bytes)}
+

{t("core.jobsTable")}

+ {metrics?.jobs.length ?
{metrics.jobs.map((job) => )}
{t("core.job")}{t("core.jobStatus")}{t("core.lastRun")}{t("core.processed")}{t("core.failed")}
{job.id}{job.status}{job.last_run_at == null ? unavailable : new Date(job.last_run_at).toLocaleString(i18n.resolvedLanguage)}{number(job.processed)}{number(job.failed)}
:

{!localSource ? t("system.adminUnavailable") : failed ? t("core.metricsUnavailable") : t("system.loading")}

} +
+
+
; +} diff --git a/apps/web/src/features/dashboard/DashboardView.css b/apps/web/src/features/dashboard/DashboardView.css index 994880b84..b2c426471 100644 --- a/apps/web/src/features/dashboard/DashboardView.css +++ b/apps/web/src/features/dashboard/DashboardView.css @@ -5,6 +5,80 @@ overflow: hidden; } +.dashboard-system { + display: grid; + gap: 16px; + min-width: 0; + margin-top: 18px; +} +.dashboard-core-metrics { padding-top: 18px; border-top: 1px solid var(--line); } + +.dashboard-system-header, +.dashboard-system-controls, +.dashboard-system-metrics { + display: flex; + align-items: center; +} + +.dashboard-system-header { + justify-content: space-between; + gap: 12px; +} + +.dashboard-system-header h2, +.dashboard-system-tables h3 { margin: 0; } +.dashboard-system-header p, +.dashboard-system-source, +.dashboard-system-tables p { margin: 4px 0 0; color: var(--fg-muted); font-size: 12px; } +.dashboard-system-controls { gap: 4px; } +.dashboard-system-controls button { + padding: 5px 9px; + border: 1px solid var(--line); + border-radius: 6px; + background: var(--surface); + color: var(--fg-muted); + cursor: pointer; +} +.dashboard-system-controls button[aria-pressed="true"] { color: var(--fg); border-color: var(--fg-muted); } +.dashboard-system-metrics { align-items: stretch; flex-wrap: wrap; gap: 8px; } +.dashboard-system-metrics > div { + display: grid; + flex: 1 1 150px; + gap: 5px; + min-width: 0; + padding: 13px; + border: 1px solid var(--line); + border-radius: 8px; + background: var(--surface); +} +.dashboard-system-metrics small, +.dashboard-system-metrics span { color: var(--fg-muted); font-size: 11px; } +.dashboard-system-metrics strong { font-size: 20px; font-variant-numeric: tabular-nums; } +.dashboard-system-tables { display: grid; grid-template-columns: repeat(2, minmax(0, 1fr)); gap: 12px; } +.dashboard-system-request-chart { + min-width: 0; + margin: 0; + padding: 14px; + border: 1px solid var(--line); + border-radius: 8px; + background: var(--surface); +} +.dashboard-system-request-chart figcaption { display: flex; align-items: baseline; justify-content: space-between; gap: 12px; font-size: 13px; } +.dashboard-system-request-chart figcaption span { color: var(--fg-muted); font-size: 11px; } +.dashboard-system-request-chart svg { display: block; width: 100%; height: 150px; margin-top: 9px; color: var(--fg); } +.dashboard-system-tables > section { min-width: 0; padding: 14px; border: 1px solid var(--line); border-radius: 8px; background: var(--surface); } +.dashboard-system-nodes { grid-column: 1 / -1; } +.dashboard-system-table-scroll { overflow-x: auto; margin-top: 10px; } +.dashboard-system-table-scroll table { width: 100%; border-collapse: collapse; font-size: 12px; } +.dashboard-system-table-scroll th, +.dashboard-system-table-scroll td { padding: 8px 7px; border-bottom: 1px solid var(--line); text-align: right; white-space: nowrap; } +.dashboard-system-table-scroll th:first-child { text-align: left; } +.dashboard-system-table-scroll thead th { color: var(--fg-muted); font-weight: 500; } +@media (max-width: 760px) { + .dashboard-system-header { align-items: flex-start; flex-direction: column; } + .dashboard-system-tables { grid-template-columns: 1fr; } +} + .dashboard-header > div:first-child { min-width: 0; } @@ -27,6 +101,34 @@ overflow-y: auto; } +.dashboard-view-tabs { + display: flex; + width: fit-content; + margin-bottom: 14px; + padding: 3px; + border: 1px solid var(--line); + border-radius: 8px; + background: var(--surface); + gap: 3px; +} + +.dashboard-view-tabs button { + padding: 7px 13px; + border: 0; + border-radius: 5px; + background: transparent; + color: var(--fg-muted); + cursor: pointer; + font-size: 12px; + font-weight: 600; +} + +.dashboard-view-tabs button[aria-selected="true"] { + background: var(--surface-raised, var(--surface)); + color: var(--fg); + box-shadow: var(--shadow-control); +} + .dashboard-overview, .dashboard-panel { min-width: 0; @@ -633,6 +735,64 @@ border-bottom: 1px solid var(--line); } +.dashboard-sandbox-insights { + padding: 16px; + border-bottom: 1px solid var(--line); +} + +.dashboard-sandbox-insights > header, +.dashboard-sandbox-diagnostics { + display: flex; + justify-content: space-between; + gap: 16px; +} + +.dashboard-sandbox-insights h3, +.dashboard-sandbox-insights h4, +.dashboard-sandbox-insights p { + margin: 0; +} + +.dashboard-sandbox-insights h3 { font-size: 14px; } +.dashboard-sandbox-insights h4 { font-size: 12px; } +.dashboard-sandbox-insights p, +.dashboard-sandbox-insights small { color: var(--fg-muted); font-size: 11px; } + +.dashboard-sandbox-insight-grid { + display: grid; + grid-template-columns: repeat(4, minmax(0, 1fr)); + gap: 8px; + margin-top: 12px; +} + +.dashboard-sandbox-insight-grid > div { + display: grid; + gap: 3px; + padding: 10px 12px; + border: 1px solid var(--line); + border-radius: 8px; + min-width: 0; +} + +.dashboard-sandbox-insight-grid strong { + font: 550 20px/1.2 var(--font-mono); + font-variant-numeric: tabular-nums; +} + +.dashboard-sandbox-insight-grid span { color: var(--fg-muted); font-size: 10px; } +.dashboard-sandbox-diagnostics { margin-top: 14px; } +.dashboard-sandbox-diagnostics > div { flex: 1; min-width: 0; } +.dashboard-sandbox-diagnostics ul { list-style: none; margin: 7px 0 0; padding: 0; } +.dashboard-sandbox-diagnostics li { display: flex; justify-content: space-between; gap: 8px; padding: 4px 0; font-size: 11px; } +.dashboard-sandbox-diagnostics li + li { border-top: 1px solid var(--line); } +.dashboard-sandbox-diagnostics li button { padding: 0; border: 0; background: none; color: var(--accent); cursor: pointer; text-align: left; } +.dashboard-sandbox-diagnostics li span { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } + +@media (max-width: 800px) { + .dashboard-sandbox-insight-grid { grid-template-columns: repeat(2, minmax(0, 1fr)); } + .dashboard-sandbox-diagnostics { flex-direction: column; } +} + .dashboard-runtime-metric { display: grid; min-width: 0; diff --git a/apps/web/src/features/dashboard/DashboardView.test.tsx b/apps/web/src/features/dashboard/DashboardView.test.tsx index 39f3a23e0..7acd92461 100644 --- a/apps/web/src/features/dashboard/DashboardView.test.tsx +++ b/apps/web/src/features/dashboard/DashboardView.test.tsx @@ -128,13 +128,14 @@ describe("Dashboard loaded-result presentation", () => { it("keeps a Runtime-only 503 scoped to the optional observation feature", () => { const html = render({ + initialTab: "observability", runtimeSnapshot: null, runtimeCollectionState: "failed", runtimeCollectionError: "Agent core request failed (503).", runtimeCollectionHasSnapshot: false, }); - expect(html).toContain("Runtime: Agent core request failed (503)."); + expect(html).toContain("Agent core request failed (503)."); expect(html).toContain("Runtime observations unavailable"); expect(html).not.toContain("Agent Core backend is not ready"); expect(html).not.toContain("Core backend is offline"); @@ -239,6 +240,7 @@ describe("Dashboard loaded-result presentation", () => { memory: { usage_bytes: 536_870_912, limit_bytes: 2_147_483_648 }, }; const html = render({ + initialTab: "observability", runtimeSnapshot: { sessions: [hosted], observations: [observation], loadedAt: 1_700_000_100_000 }, }); @@ -247,6 +249,13 @@ describe("Dashboard loaded-result presentation", () => { expect(html).toContain("Cumulative CPU / capacity"); expect(html).toContain("1m 13s / 2 cores"); expect(html).toContain("512 MiB / 2.00 GiB"); + expect(html).toContain("Sandbox diagnostics"); + expect(html).toContain("Memory ≥80%"); + const withoutMemory = render({ + initialTab: "observability", + runtimeSnapshot: { sessions: [hosted], observations: [{ ...observation, memory: { usage_bytes: null, limit_bytes: null } }], loadedAt: 1_700_000_100_000 }, + }); + expect(withoutMemory).toMatch(/Memory ≥80%<\/small>Unavailable<\/strong>/); expect(html).toContain('aria-label="Runtime live-window charts"'); expect(html).toContain("Resource trends"); expect(html).toContain("Browser-local samples · reset on reload"); @@ -355,13 +364,14 @@ describe("Dashboard loaded-result presentation", () => { memory: null, }; const html = render({ + initialTab: "observability", runtimeSnapshot: { sessions: [runtimeSession], observations: [observation], loadedAt: 1_700_000_100_000 }, runtimeCollectionState: "failed", runtimeCollectionError: "Runtime refresh failed", runtimeCollectionHasSnapshot: true, }); - expect(html).toContain("Runtime: Runtime refresh failed"); + expect(html).toContain("Runtime refresh failed"); expect(html).toContain("Live · loading history"); expect(html).not.toContain("Live · 30s"); }); diff --git a/apps/web/src/features/dashboard/DashboardView.tsx b/apps/web/src/features/dashboard/DashboardView.tsx index 5ac9a7cd2..abf058bd2 100644 --- a/apps/web/src/features/dashboard/DashboardView.tsx +++ b/apps/web/src/features/dashboard/DashboardView.tsx @@ -6,21 +6,26 @@ import { RefreshCw, Rows3, } from "lucide-react"; -import { useMemo, type ReactNode } from "react"; +import { useEffect, useMemo, useState, type ReactNode } from "react"; import { useTranslation } from "react-i18next"; -import type { AgentSession, SavedAgent } from "@agents-core-web/agents-client"; +import { AdminClient, type AgentSession, type SandboxNode, type SavedAgent } from "@agents-core-web/agents-client"; import { StatusIcon, type StatusKind } from "../../components/StatusIcon"; import { backendFailureStatus } from "../../lib/core-readiness"; +import { isLocalProxyBaseUrl } from "../../lib/connection"; +import { sandboxConsoleConfig } from "../sandbox/console-config"; import { buildDashboardSnapshot, buildRuntimeDashboardModel, + formatDashboardBytes, formatDashboardTimestamp, + formatDashboardTokens, type DashboardCollectionState, type DashboardSessionRow, } from "./dashboard-model"; import { RuntimeObservabilityContent, type RuntimeHistoryLoader } from "./RuntimeObservabilityContent"; +import { SystemObservabilityContent } from "./SystemObservabilityContent"; import type { RuntimeDashboardSnapshot } from "./runtime-snapshot"; import "./DashboardView.css"; @@ -38,6 +43,8 @@ export interface DashboardViewProps { runtimeCollectionError: string | null; runtimeCollectionHasSnapshot: boolean; loadRuntimeHistory: RuntimeHistoryLoader; + coreBaseUrl?: string; + initialTab?: "overview" | "observability"; onRefresh: () => void; onCreateAgent: () => void; onStartSession: () => void; @@ -247,6 +254,8 @@ export function DashboardView({ runtimeCollectionError, runtimeCollectionHasSnapshot, loadRuntimeHistory, + coreBaseUrl = "/v1", + initialTab = "overview", onRefresh, onCreateAgent, onStartSession, @@ -257,6 +266,25 @@ export function DashboardView({ }: DashboardViewProps) { const { t, i18n } = useTranslation("pages"); const locale = i18n.resolvedLanguage; + const [tab, setTab] = useState<"overview" | "observability">(initialTab); + const [nodes, setNodes] = useState(null); + const [nodeRefresh, setNodeRefresh] = useState(0); + useEffect(() => { + const timer = window.setInterval(() => setNodeRefresh((value) => value + 1), 30_000); + return () => window.clearInterval(timer); + }, []); + useEffect(() => { + if (!isLocalProxyBaseUrl(coreBaseUrl)) { setNodes(null); return; } + const controller = new AbortController(); + void (async () => { + const config = await sandboxConsoleConfig(controller.signal); + if (!config?.sandbox_admin) { if (!controller.signal.aborted) setNodes(null); return; } + const client = new AdminClient(); + const result = await client.listSandboxNodes({ signal: controller.signal }); + if (!controller.signal.aborted) setNodes(result.data); + })().catch(() => { if (!controller.signal.aborted) setNodes(null); }); + return () => controller.abort(); + }, [coreBaseUrl, nodeRefresh]); const snapshot = useMemo(() => buildDashboardSnapshot(agents, sessions, 6, 5), [agents, sessions]); const runtimeModel = useMemo(() => runtimeSnapshot ? buildRuntimeDashboardModel(runtimeSnapshot.sessions, runtimeSnapshot.observations) @@ -302,7 +330,7 @@ export function DashboardView({ + + + {tab === "overview" ? <>
@@ -371,16 +404,16 @@ export function DashboardView({
- - 0} - /> + + + + + node.online && node.provider_ready).length}/${nodes.length}`} detail={t("dashboard.nodeSnapshot")} />
+ : null} + {tab === "observability" ? <>
@@ -393,6 +426,7 @@ export function DashboardView({ ) : null}
+ {runtimeCollectionState === "failed" && runtimeCollectionError ?

{runtimeCollectionError}

: null} {!runtimeAvailable || !runtimeSnapshot || !runtimeModel ? (

+ + : null} + {tab === "overview" ? <>
@@ -476,6 +513,7 @@ export function DashboardView({

)}
+ : null}
); diff --git a/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx b/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx index 2c39d9d37..f40f5e62e 100644 --- a/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx +++ b/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx @@ -26,6 +26,7 @@ import { import { buildRuntimeDashboardModel, + buildSandboxInsights, formatDashboardBytes, formatDashboardDuration, formatDashboardTimestamp, @@ -312,6 +313,9 @@ export function RuntimeObservabilityContent({ [snapshot, tokenTotals], ); const summary = model.summary; + const sandbox = useMemo(() => buildSandboxInsights(model.rows), [model.rows]); + const unavailableReasons = Object.entries(sandbox.unavailableReasons) + .sort((left, right) => right[1] - left[1]); return ( <>
@@ -321,6 +325,34 @@ export function RuntimeObservabilityContent({ } label={t("runtime.metrics.tokens")} value={summary.totalTokens === null ? t("runtime.filters.unavailable") : formatDashboardTokens(summary.totalTokens, locale)} detail={t("runtime.metrics.tokenDetail", { covered: summary.tokenCoverageCount, total: summary.sessionCount })} />
+
+
+
+

{t("sandbox.title")}

+

{t("sandbox.subtitle")}

+
+ {t(stale ? "sandbox.retained" : "sandbox.current")} +
+
+
{t("sandbox.observed")}{sandbox.observed.toLocaleString(locale)}{t("sandbox.observedDetail", { total: summary.managedRuntimeCount })}
+
{t("sandbox.unavailable")}{sandbox.unavailable.toLocaleString(locale)}{t("sandbox.unavailableDetail")}
+
{t("sandbox.highMemory")}{sandbox.measuredMemory === 0 ? t("runtime.filters.unavailable") : sandbox.highMemory.toLocaleString(locale)}{t("sandbox.highMemoryDetail", { total: sandbox.measuredMemory })}
+
{t("sandbox.parked")}{(sandbox.sleeping + sandbox.pending).toLocaleString(locale)}{t("sandbox.parkedDetail", { sleeping: sandbox.sleeping, pending: sandbox.pending })}
+
+ {unavailableReasons.length > 0 || sandbox.highestMemory.length > 0 ? ( +
+
+

{t("sandbox.sampleGaps")}

+ {unavailableReasons.length ?
    {unavailableReasons.map(([reason, count]) =>
  • {t(`runtime.status.${reason}` as never)}{count}
  • )}
:

{t("sandbox.noGaps")}

} +
+
+

{t("sandbox.memoryLeaders")}

+ {sandbox.highestMemory.length ?
    {sandbox.highestMemory.map((entry) =>
  • {entry.memoryPercent.toLocaleString(locale, { maximumFractionDigits: 1 })}%
  • )}
:

{t("sandbox.noMemory")}

} +
+
+ ) : null} +
+
diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx index eeba94ed4..000f8c416 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.test.tsx @@ -47,7 +47,7 @@ describe("Runtime live-window chart accessibility", () => { })).toBe("Memory usage all series hidden; use the legend to show a series"); }); - it("renders an unavailable current value as zero without retaining a stale value", () => { + it("renders an unavailable current value as a gap without retaining a stale value", () => { const unavailable = { ...sample(120_000, null), activeSandboxCount: 0, @@ -58,13 +58,13 @@ describe("Runtime live-window chart accessibility", () => { , ); - expect(html).toContain("Runtime worker0%0"); + expect(html).toContain("Runtime workerUnavailable1"); expect(html).not.toContain("Runtime worker50%1"); - expect(html).toContain("used0 B0"); + expect(html).toContain("usedUnavailable1"); expect(html).toContain("active00"); }); - it("renders empty retained buckets as continuous zero-value chart series", () => { + it("renders empty retained buckets as missing measurements", () => { const empty = (sampledAt: number): RuntimeTrendSample => ({ ...sample(sampledAt, null), targets: [], @@ -78,13 +78,13 @@ describe("Runtime live-window chart accessibility", () => { , ); - expect(html).toContain("usage0%0"); - expect(html).toContain("used0 B0"); + expect(html).toContain("usageUnavailable2"); + expect(html).toContain("usedUnavailable2"); expect(html).toContain("active00"); - expect(html).toContain("input0/min0"); - expect(html).not.toContain("No retained CPU samples"); - expect(html).not.toContain("No complete retained memory samples"); - expect(html).not.toContain("No retained token samples"); + expect(html).toContain("inputUnavailable2"); + expect(html).toContain("No retained CPU samples"); + expect(html).toContain("No retained observed memory samples"); + expect(html).toContain("No retained token samples"); }); it("renders uPlot chart mounts and reports trends only after two real samples", () => { diff --git a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx index c85f06857..d21df864e 100644 --- a/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx +++ b/apps/web/src/features/dashboard/RuntimeTrendCharts.tsx @@ -537,13 +537,16 @@ export function RuntimeTrendCharts({ id, label: targetLabel(samples, id), tone: tones[index] ?? "blue", - points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: (sample.targets.find((target) => target.seriesId === id)?.cpuRatio ?? 0) * 100 })), + points: samples.map((sample) => { + const ratio = sample.targets.find((target) => target.seriesId === id)?.cpuRatio; + return { sampledAt: sample.sampledAt, value: ratio == null ? null : ratio * 100 }; + }), })); if (cpu.length === 0 && samples.length > 0) { - cpu.push({ id: "cpu", label: t("charts.usage"), tone: "orange", points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: 0 })) }); + cpu.push({ id: "cpu", label: t("charts.usage"), tone: "orange", points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: null })) }); } - const memoryUsed = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryUsageBytes ?? 0 })); - const memoryLimit = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryLimitBytes ?? 0 })); + const memoryUsed = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryUsageBytes })); + const memoryLimit = samples.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.memoryLimitBytes })); const active = [{ id: "active", label: t(activeDisplay === "binary" ? "charts.runtime" : "charts.active.series"), @@ -552,8 +555,8 @@ export function RuntimeTrendCharts({ points: samples.map((sample) => ({ sampledAt: sample.sampledAt, value: activeDisplay === "binary" - ? (sample.activeSandboxCount ?? 0) > 0 ? 1 : 0 - : sample.activeSandboxCount ?? 0, + ? sample.activeSandboxCount === null ? null : sample.activeSandboxCount > 0 ? 1 : 0 + : sample.activeSandboxCount, })), }] satisfies TrendSeries[]; const throughput = tokenThroughput(samples); @@ -565,8 +568,8 @@ export function RuntimeTrendCharts({ ] satisfies TrendSeries[], active, tokens: [ - { id: "input", label: t("charts.input"), tone: "orange", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.inputPerMinute ?? 0 })) }, - { id: "output", label: t("charts.output"), tone: "green", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.outputPerMinute ?? 0 })) }, + { id: "input", label: t("charts.input"), tone: "orange", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.inputPerMinute })) }, + { id: "output", label: t("charts.output"), tone: "green", points: throughput.map((sample) => ({ sampledAt: sample.sampledAt, value: sample.outputPerMinute })) }, ] satisfies TrendSeries[], }; }, [activeDisplay, samples, t]); diff --git a/apps/web/src/features/dashboard/SystemObservabilityContent.test.ts b/apps/web/src/features/dashboard/SystemObservabilityContent.test.ts new file mode 100644 index 000000000..1095d71ab --- /dev/null +++ b/apps/web/src/features/dashboard/SystemObservabilityContent.test.ts @@ -0,0 +1,24 @@ +import { describe, expect, it } from "vitest"; +import type { OperatorMetricsSummary } from "@agents-core-web/agents-client"; + +import { hasCompleteRequestCoverage } from "./SystemObservabilityContent"; + +const start = "2026-09-25T00:00:00Z"; +const next = "2026-09-25T00:01:00Z"; +const heartbeat = (time: string, dropped = 0, exportFailed = 0) => ({ + start: time, source: "request" as const, attempted_count: 0, observed_count: 0, + unavailable_count: 0, timeout_count: 0, dropped_count: dropped, export_failed_count: exportFailed, +}); +const summary = (collector: OperatorMetricsSummary["collector"]): OperatorMetricsSummary => ({ + generated_at: next, start, end: next, step_seconds: 60, + requests: [], collector, turns: [], tools: [], +}); + +describe("request coverage", () => { + it("requires every heartbeat and withholds complete-range metrics after known loss", () => { + expect(hasCompleteRequestCoverage(summary([heartbeat(start)]), 2)).toBe(false); + expect(hasCompleteRequestCoverage(summary([heartbeat(start), heartbeat(next)]), 2)).toBe(true); + expect(hasCompleteRequestCoverage(summary([heartbeat(start, 1), heartbeat(next)]), 2)).toBe(false); + expect(hasCompleteRequestCoverage(summary([heartbeat(start), heartbeat(next, 0, 1)]), 2)).toBe(false); + }); +}); diff --git a/apps/web/src/features/dashboard/SystemObservabilityContent.tsx b/apps/web/src/features/dashboard/SystemObservabilityContent.tsx new file mode 100644 index 000000000..fb7575b3f --- /dev/null +++ b/apps/web/src/features/dashboard/SystemObservabilityContent.tsx @@ -0,0 +1,144 @@ +import { useEffect, useMemo, useState } from "react"; +import { useTranslation } from "react-i18next"; +import { + AdminClient, + type OperatorMetricsRange, + type OperatorMetricsSummary, + type RequestMetricBucket, + type SandboxNode, +} from "@agents-core-web/agents-client"; + +import { isLocalProxyBaseUrl } from "../../lib/connection"; +import { formatDashboardBytes } from "./dashboard-model"; +import { SystemRequestChart } from "./SystemRequestChart"; +import { CoreServiceMetricsContent } from "./CoreServiceMetricsContent"; + +const latencyBounds = [10, 25, 50, 100, 250, 500, 1000, 2500, 5000, 10000] as const; + +function percentile95(rows: readonly RequestMetricBucket[]): number | null { + const counts = Array.from({ length: 11 }, (_, index) => rows.reduce((sum, row) => sum + (row.latency_bucket_counts[index] ?? 0), 0)); + const total = counts.reduce((sum, count) => sum + count, 0); + if (total === 0) return null; + const target = Math.ceil(total * .95); + let current = 0; + for (let index = 0; index < counts.length; index += 1) { + current += counts[index]!; + if (current >= target) return latencyBounds[index] ?? 10000; + } + return null; +} + +interface RouteSummary { + family: string; + count: number; + errors: number; + latency: number | null; +} + +function routeSummaries(rows: readonly RequestMetricBucket[]): RouteSummary[] { + const families = new Map(); + for (const row of rows) { + const previous = families.get(row.route_family) ?? []; + previous.push(row); + families.set(row.route_family, previous); + } + return [...families].map(([family, group]) => ({ + family, + count: group.reduce((sum, row) => sum + row.count, 0), + errors: group.filter((row) => row.outcome === "server_error").reduce((sum, row) => sum + row.count, 0), + latency: percentile95(group), + })).sort((left, right) => right.count - left.count); +} + +export function hasCompleteRequestCoverage(summary: OperatorMetricsSummary | null, expectedBuckets: number): boolean { + if (summary === null) return false; + const requestCollector = summary.collector.filter((row) => row.source === "request"); + return new Set(requestCollector.map((row) => row.start)).size >= expectedBuckets && + requestCollector.every((row) => row.dropped_count === 0 && row.export_failed_count === 0); +} + +export function SystemObservabilityContent({ coreBaseUrl, nodes }: { coreBaseUrl: string; nodes: SandboxNode[] | null }) { + const { t, i18n } = useTranslation("dashboard"); + const locale = i18n.resolvedLanguage; + const [range, setRange] = useState("1h"); + const [refresh, setRefresh] = useState(0); + const [summary, setSummary] = useState(null); + const [state, setState] = useState<"loading" | "ready" | "unavailable" | "failed">("loading"); + useEffect(() => { + const timer = window.setInterval(() => setRefresh((value) => value + 1), 30_000); + return () => window.clearInterval(timer); + }, []); + useEffect(() => { + const controller = new AbortController(); + if (!isLocalProxyBaseUrl(coreBaseUrl)) { + setSummary(null); + setState("unavailable"); + return () => controller.abort(); + } + setState("loading"); + setSummary(null); + void (async () => { + const metricsClient = new AdminClient(); + const metrics = await metricsClient.retrieveObservability(range, { signal: controller.signal }); + if (controller.signal.aborted) return; + setSummary(metrics); + setState("ready"); + })().catch(() => { if (!controller.signal.aborted) setState("failed"); }); + return () => controller.abort(); + }, [coreBaseUrl, range, refresh]); + + const routes = useMemo(() => routeSummaries(summary?.requests ?? []), [summary]); + const total = routes.reduce((sum, row) => sum + row.count, 0); + const serverErrors = routes.reduce((sum, row) => sum + row.errors, 0); + const p95 = percentile95(summary?.requests ?? []); + const durationMinutes = range === "1h" ? 60 : range === "6h" ? 360 : 1440; + const expectedBuckets = durationMinutes / ((summary?.step_seconds ?? 60) / 60); + const requestCollector = (summary?.collector ?? []).filter((row) => row.source === "request"); + const heartbeatBuckets = new Set(requestCollector.map((row) => row.start)).size; + const completeRequestCoverage = hasCompleteRequestCoverage(summary, expectedBuckets); + const requestRate = completeRequestCoverage ? total / durationMinutes : null; + const errorRate = completeRequestCoverage && total > 0 ? serverErrors / total * 100 : null; + const collector = summary?.collector ?? []; + const dropped = collector.reduce((sum, row) => sum + row.dropped_count + row.export_failed_count, 0); + const onlineNodes = nodes?.filter((node) => node.online && node.provider_ready).length; + const nodeCapacity = nodes?.reduce((sum, node) => sum + node.max_active, 0); + const number = (value: number) => value.toLocaleString(locale); + const unavailable = t("system.unavailableValue"); + + return
+
+

{t("system.title")}

{t("system.subtitle")}

+
+ {(["1h", "6h", "24h"] as const).map((option) => )} +
+
+

{state === "loading" ? t("system.loading") : state === "failed" ? t("system.stale") : state === "unavailable" ? t("system.adminUnavailable") : t("system.source")}

+
+
{t("system.requestRate")}{requestRate === null ? unavailable : `${number(Math.round(requestRate))}/min`}{t("system.requestCount", { count: total })}
+
{t("system.serverErrorRate")}{errorRate === null ? unavailable : `${errorRate.toFixed(1)}%`}{t("system.serverErrorCount", { count: serverErrors })}
+
{t("system.latencyP95")}{!completeRequestCoverage || p95 === null ? unavailable : p95 === 10000 ? ">10 s" : `${number(p95)} ms`}{t("system.histogramEstimate")}
+
{t("system.collectorCoverage")}{summary === null ? unavailable : `${number(heartbeatBuckets)}/${number(expectedBuckets)}`}{t("system.collectorLoss", { count: dropped })}
+
{t("system.nodes")}{nodes === null ? unavailable : `${number(onlineNodes ?? 0)}/${number(nodes.length)}`}{nodeCapacity === undefined ? unavailable : t("system.nodeCapacity", { count: nodeCapacity })}
+
+ +
+

{t("system.routeTable")}

{t("system.coveredBuckets", { covered: heartbeatBuckets, total: expectedBuckets })}

+ {routes.length ?
{routes.map((row) => )}
{t("system.route")}{t("system.requests")}{t("system.errors")}{t("system.p95")}
{row.family}{number(row.count)}{number(row.errors)}{row.latency === null ? unavailable : row.latency === 10000 ? ">10 s" : `${number(row.latency)} ms`}
:

{t("system.noRequestSamples")}

} +
+

{t("system.collectorTable")}

+ {collector.length ?
{[...new Set(collector.map((row) => row.source))].map((source) => { const rows = collector.filter((row) => row.source === source); return ; })}
{t("system.sourceName")}{t("system.attempted")}{t("system.observed")}{t("system.lost")}
{source}{number(rows.reduce((sum, row) => sum + row.attempted_count, 0))}{number(rows.reduce((sum, row) => sum + row.observed_count, 0))}{number(rows.reduce((sum, row) => sum + row.dropped_count + row.export_failed_count, 0))}
:

{t("system.noCollectorSamples")}

} +
+

{t("system.turnTable")}

+ {summary?.turns.length ?
{[...new Set(summary.turns.map((row) => row.status))].map((status) => { const rows = summary.turns.filter((row) => row.status === status); const count = rows.reduce((sum, row) => sum + row.count, 0); const queue = rows.reduce((values, row) => row.queue_p95_ms === null ? values : [...values, row.queue_p95_ms], []); const execution = rows.reduce((values, row) => row.execution_p95_ms === null ? values : [...values, row.execution_p95_ms], []); return ; })}
{t("system.turnStatus")}{t("system.turnCount")}{t("system.queueP95")}{t("system.executionP95")}
{status}{number(count)}{queue.length ? `${number(Math.round(Math.max(...queue)))} ms` : unavailable}{execution.length ? `${number(Math.round(Math.max(...execution)))} ms` : unavailable}
:

{t("system.noTurns")}

} +
+

{t("system.toolTable")}

+ {summary?.tools.length ?
{summary.tools.map((row, index) => )}
{t("system.toolCategory")}{t("system.toolOutcome")}{t("system.toolCount")}{t("system.toolTiming")}
{row.category}{row.outcome}{number(row.count)}{row.duration_p95_ms === null ? unavailable : `${number(Math.round(row.duration_p95_ms))} ms · ${number(row.timed_count)}/${number(row.count)}`}
:

{t("system.noToolAttempts")}

} +
+

{t("system.modelTable")}

{t("system.modelUnavailable")}

+

{t("system.nodeTable")}

+ {nodes?.length ?
{nodes.map((node) => )}
{t("system.node")}{t("system.status")}{t("system.active")}{t("system.retained")}{t("system.cores")}{t("system.availableMemory")}
{node.name}{t(node.online && node.provider_ready ? "system.online" : "system.offline")}{number(node.active)} / {number(node.max_active)}{number(node.retained)}{node.cpu_count === null ? unavailable : number(node.cpu_count)}{node.available_memory_bytes === null ? unavailable : formatDashboardBytes(node.available_memory_bytes)}
:

{nodes === null ? t("system.nodeUnavailable") : t("system.noNodes")}

} +
+
+ +
; +} diff --git a/apps/web/src/features/dashboard/SystemRequestChart.tsx b/apps/web/src/features/dashboard/SystemRequestChart.tsx new file mode 100644 index 000000000..1822ff44e --- /dev/null +++ b/apps/web/src/features/dashboard/SystemRequestChart.tsx @@ -0,0 +1,53 @@ +import { useMemo } from "react"; +import { useTranslation } from "react-i18next"; +import type { OperatorMetricsSummary } from "@agents-core-web/agents-client"; + +export function SystemRequestChart({ summary }: { summary: OperatorMetricsSummary | null }) { + const { t, i18n } = useTranslation("dashboard"); + const points = useMemo(() => { + if (!summary) return []; + const start = Date.parse(summary.start); + const end = Date.parse(summary.end); + const stepMS = summary.step_seconds * 1_000; + const requests = new Map(); + for (const row of summary.requests) { + const at = Date.parse(row.start); + const prior = requests.get(at) ?? { total: 0, errors: 0 }; + prior.total += row.count; + if (row.outcome === "server_error") prior.errors += row.count; + requests.set(at, prior); + } + const covered = new Set(summary.collector.filter((row) => row.source === "request").map((row) => Date.parse(row.start))); + const result = []; + for (let at = start; at < end; at += stepMS) { + result.push({ at, value: covered.has(at) ? requests.get(at) ?? { total: 0, errors: 0 } : null }); + } + return result; + }, [summary]); + const known = points.filter((point) => point.value !== null); + const maximum = Math.max(1, ...known.map((point) => point.value?.total ?? 0)); + const width = 780; + const height = 150; + const left = 28; + const baseline = 124; + const plotWidth = width - left - 8; + const slot = plotWidth / Math.max(1, points.length); + return
+
{t("system.requestTrend")}{t("system.requestTrendDetail", { covered: known.length, total: points.length })}
+ {known.length ? + + {points.map((point, index) => { + if (point.value === null) return null; + const x = left + index * slot + slot * .12; + const barWidth = Math.max(1, slot * .75); + const barHeight = Math.max(1, point.value.total / maximum * 100); + const errorHeight = Math.max(0, point.value.errors / maximum * 100); + return + {`${new Date(point.at).toLocaleString(i18n.resolvedLanguage)}: ${point.value.total} ${t("system.requests")}, ${point.value.errors} ${t("system.errors")}`} + + {errorHeight > 0 ? : null} + ; + })} + :

{t("system.noRequestSamples")}

} +
; +} diff --git a/apps/web/src/features/dashboard/dashboard-model.test.ts b/apps/web/src/features/dashboard/dashboard-model.test.ts index 349e6b3ae..8288f1698 100644 --- a/apps/web/src/features/dashboard/dashboard-model.test.ts +++ b/apps/web/src/features/dashboard/dashboard-model.test.ts @@ -5,6 +5,7 @@ import type { AgentSession, RuntimeObservation, SavedAgent, TokenUsage } from "@ import { buildDashboardSnapshot, buildRuntimeDashboardModel, + buildSandboxInsights, dashboardEnvironmentLabel, dashboardEnvironmentProfile, dashboardStatusLabel, @@ -355,6 +356,56 @@ describe("Dashboard loaded-snapshot model", () => { totalTokens: 26, tokenCoverageCount: 2, }); + expect(buildSandboxInsights(model.rows)).toMatchObject({ + observed: 1, + unavailable: 0, + highMemory: 0, + measuredMemory: 1, + highestMemory: [{ sessionId: second.id, memoryPercent: 37.5 }], + }); + }); + + it("separates sleeping allocations from unexpected sampling gaps", () => { + const base = session("11111111-1111-4111-8111-111111111111"); + const observation: RuntimeObservation = { + id: base.id, object: "agent.runtime_observation", session_id: base.id, + environment_id: "33333333-3333-4333-8333-333333333333", mode: "openai_hosted", + provider_type: "docker", + instance: { kind: "managed_allocation", allocation_id: "allocation-1", device_id: null, connection_generation: null }, + lifecycle_state: "sleeping", status: "unavailable", reason: "runtime_not_running", + allocation_created_at: 100, resolved_at: 220, observed_at: null, started_at: null, + cpu: null, memory: null, + }; + const failed = { ...observation, id: "other", session_id: "other", instance: { ...observation.instance, allocation_id: "allocation-2" }, lifecycle_state: "active" as const, reason: "sample_timeout" as const }; + const rows = buildRuntimeDashboardModel([base, session("other")], [observation, failed]).rows; + expect(buildSandboxInsights(rows)).toMatchObject({ + sleeping: 1, pending: 0, unavailable: 1, + unavailableReasons: { sample_timeout: 1 }, + }); + }); + + it("does not pick a conflicting same-second allocation observation by list order", () => { + const first = session("first"); + const second = session("second"); + const base: RuntimeObservation = { + id: first.id, object: "agent.runtime_observation", session_id: first.id, + environment_id: "33333333-3333-4333-8333-333333333333", mode: "openai_hosted", + provider_type: "docker", instance: { kind: "managed_allocation", allocation_id: "allocation-1", device_id: null, connection_generation: null }, + lifecycle_state: "active", status: "observed", reason: null, + allocation_created_at: 100, resolved_at: 220, observed_at: 210, started_at: 150, + cpu: null, memory: { usage_bytes: 900, limit_bytes: 1000 }, + }; + const conflict: RuntimeObservation = { + ...base, id: second.id, session_id: second.id, status: "unavailable", reason: "sample_timeout", + observed_at: null, started_at: null, cpu: null, memory: null, + }; + const insights = (observations: RuntimeObservation[]) => buildSandboxInsights(buildRuntimeDashboardModel([first, second], observations).rows); + for (const order of [[base, conflict], [conflict, base]]) { + expect(insights(order)).toMatchObject({ + observed: 0, unavailable: 1, highMemory: 0, measuredMemory: 0, + unavailableReasons: { sample_unavailable: 1 }, + }); + } }); it("holds each Session's last reported tokens in the summary while public usage is null", () => { diff --git a/apps/web/src/features/dashboard/dashboard-model.ts b/apps/web/src/features/dashboard/dashboard-model.ts index 9f506aef7..7fd967555 100644 --- a/apps/web/src/features/dashboard/dashboard-model.ts +++ b/apps/web/src/features/dashboard/dashboard-model.ts @@ -81,6 +81,26 @@ export interface RuntimeDashboardModel { rows: RuntimeDashboardRow[]; } +export interface SandboxInsight { + allocationId: string; + sessionId: string; + title: string; + memoryPercent: number; + memoryUsageBytes: number; + memoryLimitBytes: number; +} + +export interface SandboxInsights { + observed: number; + unavailable: number; + pending: number; + sleeping: number; + highMemory: number; + measuredMemory: number; + unavailableReasons: Partial, number>>; + highestMemory: SandboxInsight[]; +} + const sessionStatuses = new Set([ "idle", "in_progress", @@ -475,6 +495,75 @@ export function buildRuntimeDashboardModel( }; } +/** Current managed allocations only; duplicate Session observations share one allocation. */ +export function buildSandboxInsights(rows: readonly RuntimeDashboardRow[]): SandboxInsights { + const byAllocation = new Map(); + let unallocatedPending = 0; + for (const row of rows) { + const allocationId = row.observation.instance.allocation_id; + if (row.observation.mode !== "openai_hosted") continue; + if (!allocationId) { + if (row.observation.lifecycle_state === "pending") unallocatedPending += 1; + continue; + } + const previous = byAllocation.get(allocationId); + if (previous) previous.push(row); + else byAllocation.set(allocationId, [row]); + } + const insights: SandboxInsights = { + observed: 0, unavailable: 0, pending: unallocatedPending, sleeping: 0, + highMemory: 0, measuredMemory: 0, unavailableReasons: {}, highestMemory: [], + }; + for (const [allocationId, candidates] of byAllocation) { + const newestTime = candidates.reduce((latest, candidate) => Math.max(latest, candidate.observation.resolved_at), 0); + const newest = candidates.filter((candidate) => candidate.observation.resolved_at === newestTime); + const signature = (candidate: RuntimeDashboardRow) => { + const value = candidate.observation; + return JSON.stringify([value.lifecycle_state, value.status, value.reason, value.observed_at, value.memory]); + }; + // Public timestamps have second resolution. Conflicting observations in one + // second have no reliable order, so do not publish a pressure measurement. + if (new Set(newest.map(signature)).size > 1) { + insights.unavailable += 1; + insights.unavailableReasons.sample_unavailable = (insights.unavailableReasons.sample_unavailable ?? 0) + 1; + continue; + } + const row = newest.sort((left, right) => left.observation.session_id.localeCompare(right.observation.session_id))[0]!; + const observation = row.observation; + if (observation.lifecycle_state === "pending") insights.pending += 1; + if (observation.lifecycle_state === "sleeping") insights.sleeping += 1; + if (observation.status === "unavailable") { + // Parked and stopped allocations are not sampling failures. + if (observation.lifecycle_state !== "sleeping" && observation.lifecycle_state !== "pending" && observation.lifecycle_state !== "stopped") { + insights.unavailable += 1; + if (observation.reason) { + insights.unavailableReasons[observation.reason] = (insights.unavailableReasons[observation.reason] ?? 0) + 1; + } + } + continue; + } + if (observation.status !== "observed") continue; + insights.observed += 1; + const usage = safeNonNegativeInteger(observation.memory?.usage_bytes); + const limit = safeNonNegativeInteger(observation.memory?.limit_bytes); + if (usage === null || limit === null || limit === 0) continue; + insights.measuredMemory += 1; + const memoryPercent = usage / limit * 100; + if (memoryPercent >= 80) insights.highMemory += 1; + insights.highestMemory.push({ + allocationId, + sessionId: row.observation.session_id, + title: row.session.title, + memoryPercent, + memoryUsageBytes: usage, + memoryLimitBytes: limit, + }); + } + insights.highestMemory.sort((left, right) => right.memoryPercent - left.memoryPercent || left.allocationId.localeCompare(right.allocationId)); + insights.highestMemory = insights.highestMemory.slice(0, 5); + return insights; +} + export function formatDashboardBytes(value: number | null): string { if (value === null) return "Unavailable"; const units = ["B", "KiB", "MiB", "GiB", "TiB"]; diff --git a/apps/web/src/i18n/locales/en/dashboard.ts b/apps/web/src/i18n/locales/en/dashboard.ts index 449e5a735..06e280044 100644 --- a/apps/web/src/i18n/locales/en/dashboard.ts +++ b/apps/web/src/i18n/locales/en/dashboard.ts @@ -1,4 +1,42 @@ export const dashboard = { + system: { + title: "System observability", subtitle: "Completed HTTP requests, collection coverage, and sandbox nodes", + range: "System metrics range", loading: "Loading operator metrics…", stale: "Operator metrics unavailable · retrying", + adminUnavailable: "Deployment administrator metrics are unavailable on this connection.", source: "Core operator metrics · completed time buckets", + unavailableValue: "Unavailable", requestRate: "Request rate", requestCount: "{{count}} observed completed requests", + serverErrorRate: "HTTP 5xx rate", serverErrorCount: "{{count}} observed server errors", latencyP95: "Latency p95", histogramEstimate: "Estimated from fixed latency buckets", + collectorCoverage: "Request collector buckets", collectorLoss: "{{count}} dropped or failed writes", nodes: "Ready nodes", nodeCapacity: "{{count}} active slots configured", + routeTable: "Request routes", coveredBuckets: "Observed buckets: {{covered}}/{{total}}", route: "Route family", requests: "Requests", errors: "5xx", p95: "p95", + requestTrend: "Request volume", requestTrendDetail: "{{covered}}/{{total}} buckets covered · gaps are unavailable", requestTrendLabel: "Request volume over time; {{covered}} of {{total}} buckets covered", + collectorTable: "Collector health", sourceName: "Source", attempted: "Attempted", observed: "Observed", lost: "Dropped / failed", + turnTable: "Terminal Turns", turnStatus: "Outcome", turnCount: "Turns", queueP95: "Worst bucket queue p95", executionP95: "Worst bucket run p95", noTurns: "No terminal Turns in this range.", + toolTable: "Actual tool attempts", toolCategory: "Category", toolOutcome: "Outcome", toolCount: "Attempts", toolTiming: "Bucket p95 · timed coverage", noToolAttempts: "No terminal tool attempt observations in this range.", + modelTable: "Actual model invocations", modelUnavailable: "Invocation-level collection is not yet available from every harness. Configured models and Turn usage are not counted as calls.", + nodeTable: "Sandbox nodes", node: "Node", status: "Status", active: "Active / capacity", retained: "Retained", cores: "CPU cores", availableMemory: "Available memory", online: "Ready", offline: "Unavailable", + noRequestSamples: "No request buckets collected in this range.", noCollectorSamples: "No collector coverage recorded in this range.", noNodes: "No sandbox nodes registered.", nodeUnavailable: "Node data unavailable.", + }, + core: { + title: "Core service", subtitle: "Execution queue, database, background jobs, and process state", + status: "Service: {{status}}", metricsUnavailable: "Core service metrics unavailable.", + slots: "Execution slots", executionOwner: "Execution owner: {{status}}", yes: "Yes", no: "No", + queue: "Queued Turns", runningCount: "{{value}} running", databasePing: "Database ping p95", + pool: "Pool {{used}}/{{max}} in use", processMemory: "Process memory", goroutines: "{{value}} goroutines", + executionTable: "Execution queue", queued: "Queued", running: "In progress", waitingDaemon: "Waiting for daemon", + connectedDaemons: "Connected daemons", oldestQueue: "Oldest queued", queueP95: "Queue wait p95", + interrupted: "Interrupted", unavailable: "Execution unavailable", databaseTable: "Database and process", + pingP50: "Ping p50", pingP95: "Ping p95", poolInUse: "Pool in use", poolIdle: "Pool idle", databaseSize: "Database size", + jobsTable: "Background jobs", job: "Job", jobStatus: "Status", lastRun: "Last run", processed: "Processed", failed: "Failed", + }, + sandbox: { + title: "Sandbox diagnostics", subtitle: "Current allocation identity, sampling coverage, and memory pressure", + current: "Current complete snapshot", retained: "Last complete snapshot · refresh failed", + observed: "Observed", observedDetail: "of {{total}} managed targets", + unavailable: "Sampling gaps", unavailableDetail: "Excludes sleeping, pending and stopped", + highMemory: "Memory ≥80%", highMemoryDetail: "of {{total}} with measured usage and limit", + parked: "Parked", parkedDetail: "{{sleeping}} sleeping · {{pending}} pending", + sampleGaps: "Why samples are missing", noGaps: "No unexpected sampling gaps", + memoryLeaders: "Highest memory pressure", noMemory: "No measured memory and limit", + }, runtime: { explorer: "Runtime target explorer", targets: "Runtime targets", resourceSnapshot: "Runtime resource snapshot", search: "Search Runtime targets", searchPlaceholder: "Search Session, provider, or identity", visible: "{{value}} visible", diff --git a/apps/web/src/i18n/locales/en/pages.ts b/apps/web/src/i18n/locales/en/pages.ts index 9c6de3a3d..be3773f72 100644 --- a/apps/web/src/i18n/locales/en/pages.ts +++ b/apps/web/src/i18n/locales/en/pages.ts @@ -1,5 +1,9 @@ export const pages = { dashboard: { + viewTabs: "Dashboard views", overviewTab: "Overview", observabilityTab: "Observability", + activeSandboxes: "Active sandboxes", currentRuntimeSnapshot: "Current Runtime snapshot", totalTokens: "Reported tokens", reportedUsage: "Session usage with coverage", + cpuCapacity: "Sandbox CPU capacity", measuredCores: "Cores with a current observation", memoryLimit: "Sandbox memory limit", measuredLimit: "Limits with a current observation", + readyNodes: "Ready nodes", nodeSnapshot: "Deployment node heartbeat", title: "Dashboard", subtitle: "Agents and Sessions that may need your attention.", refresh: "Refresh", refreshing: "Refreshing…", refreshLabel: "Refresh Dashboard snapshot", snapshotIncomplete: "Snapshot incomplete", snapshotStale: "Using the last successful snapshot", snapshotRefreshing: "Refreshing snapshot", snapshotReady: "Snapshot ready", snapshotDetail: "Latest complete paginated reads · not a live Core total or runtime-readiness signal", dataSources: "Dashboard data sources", agents: "Agents", sessions: "Sessions", runtime: "Runtime", unavailable: "Unavailable", savedDefinitions: "Saved definitions", inSnapshot: "In this snapshot", inProgress: "In progress", reportedStatus: "Core-reported status", needsAttention: "Needs attention", attentionDetail: "Requires action or failed", diff --git a/apps/web/src/i18n/locales/zh-CN/dashboard.ts b/apps/web/src/i18n/locales/zh-CN/dashboard.ts index 995f1720b..0fe4937a6 100644 --- a/apps/web/src/i18n/locales/zh-CN/dashboard.ts +++ b/apps/web/src/i18n/locales/zh-CN/dashboard.ts @@ -1,4 +1,42 @@ export const dashboard = { + system: { + title: "系统观测", subtitle: "已完成的 HTTP 请求、采集覆盖与 Sandbox 节点", + range: "系统指标时间范围", loading: "正在加载管理员指标…", stale: "管理员指标不可用 · 正在重试", + adminUnavailable: "当前连接无法读取部署管理员指标。", source: "Core 管理员指标 · 已完成的时间桶", + unavailableValue: "不可用", requestRate: "请求速率", requestCount: "已观测 {{count}} 个完成请求", + serverErrorRate: "HTTP 5xx 比例", serverErrorCount: "已观测 {{count}} 个服务端错误", latencyP95: "延迟 p95", histogramEstimate: "按固定延迟桶估算", + collectorCoverage: "请求采集时间桶", collectorLoss: "{{count}} 次丢弃或写入失败", nodes: "就绪节点", nodeCapacity: "配置 {{count}} 个活动名额", + routeTable: "请求分类", coveredBuckets: "已观测时间桶:{{covered}}/{{total}}", route: "路由分类", requests: "请求数", errors: "5xx", p95: "p95", + requestTrend: "请求量", requestTrendDetail: "覆盖 {{covered}}/{{total}} 个时间桶 · 缺口表示不可用", requestTrendLabel: "请求量趋势;覆盖 {{covered}}/{{total}} 个时间桶", + collectorTable: "采集健康", sourceName: "来源", attempted: "尝试", observed: "成功", lost: "丢弃/失败", + turnTable: "终态 Turn", turnStatus: "结果", turnCount: "Turn 数", queueP95: "最差时间桶排队 p95", executionP95: "最差时间桶执行 p95", noTurns: "该时段没有终态 Turn。", + toolTable: "实际工具调用", toolCategory: "类别", toolOutcome: "结果", toolCount: "调用数", toolTiming: "时间桶 p95 · 时长覆盖", noToolAttempts: "该时段没有终态工具调用观测。", + modelTable: "实际模型调用", modelUnavailable: "各 Harness 尚未提供完整的单次调用采集。不会将配置模型或 Turn 用量当作调用次数。", + nodeTable: "Sandbox 节点", node: "节点", status: "状态", active: "活动/容量", retained: "保留", cores: "CPU 核数", availableMemory: "可用内存", online: "就绪", offline: "不可用", + noRequestSamples: "该时段尚无请求时间桶。", noCollectorSamples: "该时段尚无采集覆盖记录。", noNodes: "尚未注册 Sandbox 节点。", nodeUnavailable: "节点数据不可用。", + }, + core: { + title: "Core 服务", subtitle: "执行队列、数据库、后台任务与进程状态", + status: "服务状态:{{status}}", metricsUnavailable: "Core 服务指标不可用。", + slots: "执行名额", executionOwner: "执行所有权:{{status}}", yes: "已取得", no: "未取得", + queue: "排队 Turn", runningCount: "运行中 {{value}}", databasePing: "数据库 Ping p95", + pool: "连接池占用 {{used}}/{{max}}", processMemory: "进程内存", goroutines: "{{value}} 个 Goroutine", + executionTable: "执行队列", queued: "排队中", running: "运行中", waitingDaemon: "等待 Daemon", + connectedDaemons: "已连接 Daemon", oldestQueue: "最长排队", queueP95: "排队等待 p95", + interrupted: "执行中断", unavailable: "执行不可用", databaseTable: "数据库与进程", + pingP50: "Ping p50", pingP95: "Ping p95", poolInUse: "连接池占用", poolIdle: "连接池空闲", databaseSize: "数据库大小", + jobsTable: "后台任务", job: "任务", jobStatus: "状态", lastRun: "上次运行", processed: "已处理", failed: "失败", + }, + sandbox: { + title: "Sandbox 诊断", subtitle: "按分配身份查看采样覆盖、生命周期与内存压力", + current: "当前完整快照", retained: "上次完整快照 · 刷新失败", + observed: "已观测", observedDetail: "共 {{total}} 个托管目标", + unavailable: "采样缺口", unavailableDetail: "不含休眠、待分配和已停止状态", + highMemory: "内存 ≥80%", highMemoryDetail: "共 {{total}} 个具备用量和上限采样", + parked: "暂停/等待", parkedDetail: "{{sleeping}} 个休眠 · {{pending}} 个等待分配", + sampleGaps: "采样缺失原因", noGaps: "没有意外采样缺口", + memoryLeaders: "内存压力最高", noMemory: "暂无内存用量及上限采样", + }, runtime: { explorer: "Runtime 目标浏览器", targets: "Runtime 目标", resourceSnapshot: "Runtime 资源快照", search: "搜索 Runtime 目标", searchPlaceholder: "搜索会话、Provider 或标识", visible: "显示 {{value}} 项", diff --git a/apps/web/src/i18n/locales/zh-CN/pages.ts b/apps/web/src/i18n/locales/zh-CN/pages.ts index 0a1bdc8c5..9a703e8eb 100644 --- a/apps/web/src/i18n/locales/zh-CN/pages.ts +++ b/apps/web/src/i18n/locales/zh-CN/pages.ts @@ -1,5 +1,9 @@ export const pages = { dashboard: { + viewTabs: "大盘视图", overviewTab: "概览", observabilityTab: "观测大盘", + activeSandboxes: "活跃 Sandbox", currentRuntimeSnapshot: "当前 Runtime 快照", totalTokens: "已报告 Token", reportedUsage: "有覆盖的会话用量", + cpuCapacity: "Sandbox CPU 容量", measuredCores: "有当前观测的核数", memoryLimit: "Sandbox 内存上限", measuredLimit: "有当前观测的上限", + readyNodes: "就绪节点", nodeSnapshot: "部署节点心跳", title: "概览", subtitle: "查看可能需要关注的智能体和会话。", refresh: "刷新", refreshing: "刷新中…", refreshLabel: "刷新概览快照", snapshotIncomplete: "快照不完整", snapshotStale: "正在使用最近一次成功快照", snapshotRefreshing: "正在刷新快照", snapshotReady: "快照已就绪", snapshotDetail: "最近一次完整分页读取 · 不代表 Core 实时总量或运行时就绪", dataSources: "概览数据源", agents: "智能体", sessions: "会话", runtime: "运行时", unavailable: "不可用", savedDefinitions: "已保存定义", inSnapshot: "当前快照", inProgress: "进行中", reportedStatus: "Core 报告状态", needsAttention: "需要关注", attentionDetail: "需要操作或已失败", diff --git a/apps/web/vite.config.ts b/apps/web/vite.config.ts index 6ec76098e..d8ee4b8a2 100644 --- a/apps/web/vite.config.ts +++ b/apps/web/vite.config.ts @@ -46,6 +46,7 @@ export default defineConfig(({ command, mode }) => { server: { proxy: { "/core/v1/sandbox": { target, changeOrigin: true }, + "/core/v1/admin": { target, changeOrigin: true }, "/v1": { target, changeOrigin: true, diff --git a/contracts/agents-api/runtime-observability.md b/contracts/agents-api/runtime-observability.md index 143c45b3b..1962abb51 100644 --- a/contracts/agents-api/runtime-observability.md +++ b/contracts/agents-api/runtime-observability.md @@ -83,9 +83,20 @@ Web projects active Runtime state differently by scope. The Dashboard shows one summed series of distinct allocation identities: live snapshots count `lifecycle_state: active`, while retained buckets count successfully observed allocations because lifecycle state is not retained yet. The single-Session view -collapses the same value to `1` or `0`. Missing or unavailable retained values are -currently rendered as zero, so this presentation intentionally does not yet -distinguish sleeping from collection failure. +collapses the same value to `1` or `0`. Empty retained buckets have no active +observation; an observed zero in a current lifecycle snapshot remains zero. +Retained history does not distinguish sleeping from collection failure because it +does not retain lifecycle state. + +The Dashboard's current Sandbox diagnostics deduplicate managed observations by +allocation ID, report sampling gaps separately from sleeping, pending and stopped +allocations, and rank measured memory usage against a known limit. A target with +missing usage or limit has no pressure percentage. Live and retained chart gaps +remain null rather than fabricated zero; an observed lifecycle count of zero is +still zero. Conflicting observations of one allocation with the same public +second-resolution timestamp count as an ambiguous sample gap. The broader +request, model and tool collection plan is documented in +[system observability](system-observability-plan.md). Future automatic suspension requires a separate durable control model, including an activity revision and timestamps such as `idle_since` and diff --git a/contracts/agents-api/sandbox-manager.openapi.yaml b/contracts/agents-api/sandbox-manager.openapi.yaml index 2fbec10e1..473cd22d5 100644 --- a/contracts/agents-api/sandbox-manager.openapi.yaml +++ b/contracts/agents-api/sandbox-manager.openapi.yaml @@ -96,6 +96,33 @@ definitions: rotate: type: boolean type: object + api.OperatorMetricsResponse: + properties: + collector: + items: + $ref: '#/definitions/store.CollectorMetricBucket' + type: array + end: + type: string + generated_at: + type: string + requests: + items: + $ref: '#/definitions/store.RequestMetricBucket' + type: array + start: + type: string + step_seconds: + type: integer + tools: + items: + $ref: '#/definitions/store.ToolMetricBucket' + type: array + turns: + items: + $ref: '#/definitions/store.TurnMetricBucket' + type: array + type: object api.ProjectAPIKeyRequest: properties: name: @@ -419,6 +446,25 @@ definitions: revoked_at: type: string type: object + store.CollectorMetricBucket: + properties: + attempted_count: + type: integer + dropped_count: + type: integer + export_failed_count: + type: integer + observed_count: + type: integer + source: + type: string + start: + type: string + timeout_count: + type: integer + unavailable_count: + type: integer + type: object store.CopyAssetsResult: properties: mappings: @@ -502,6 +548,23 @@ definitions: has_more: type: boolean type: object + store.RequestMetricBucket: + properties: + count: + type: integer + latency_bucket_counts: + items: + type: integer + type: array + latency_sum_ms: + type: integer + outcome: + type: string + route_family: + type: string + start: + type: string + type: object store.ResourceOwner: properties: admin_audit_id: @@ -667,6 +730,34 @@ definitions: maintenance: type: boolean type: object + store.ToolMetricBucket: + properties: + category: + type: string + count: + type: integer + duration_p95_ms: + type: number + outcome: + type: string + start: + type: string + timed_count: + type: integer + type: object + store.TurnMetricBucket: + properties: + count: + type: integer + execution_p95_ms: + type: number + queue_p95_ms: + type: number + start: + type: string + status: + type: string + type: object store.WriteOperation: properties: action: @@ -2836,6 +2927,42 @@ paths: summary: Retrieve Core operational metrics tags: - Core Administration + /core/v1/admin/observability: + get: + description: Deployment administrator only. Returns fixed-range aggregate requests and collector coverage without paths, payloads or project identifiers. + parameters: + - description: 1h, 6h, or 24h (default 1h) + in: query + name: range + type: string + produces: + - application/json + responses: + "200": + description: OK + schema: + $ref: '#/definitions/api.OperatorMetricsResponse' + "400": + description: Bad Request + schema: + $ref: '#/definitions/v1.ErrorResponse' + "401": + description: Unauthorized + schema: + $ref: '#/definitions/v1.ErrorResponse' + "500": + description: Internal Server Error + schema: + $ref: '#/definitions/v1.ErrorResponse' + "503": + description: Service Unavailable + schema: + $ref: '#/definitions/v1.ErrorResponse' + security: + - DeploymentAdminAuth: [] + summary: Retrieve bounded operator observability metrics + tags: + - Observability /core/v1/admin/projects: get: parameters: diff --git a/contracts/agents-api/system-observability-plan.md b/contracts/agents-api/system-observability-plan.md new file mode 100644 index 000000000..801a53900 --- /dev/null +++ b/contracts/agents-api/system-observability-plan.md @@ -0,0 +1,99 @@ +# System observability collection plan + +Status: implementation inventory for the operator Dashboard. Runtime history, +request buckets, terminal Turns, tool attempts, and node state have independent +sources. The model-attempt table is reserved but has no writer; its panel must +remain unavailable until native per-attempt events are qualified. + +## Existing evidence + +| Signal | Source | Current coverage | Dashboard meaning | +| --- | --- | --- | --- | +| Sandbox lifecycle | Core allocation state | Managed Docker and microsandbox | Current control state, not execution success | +| CPU time and capacity | Provider observation | Current and periodic history when sampling is available | Cumulative time; utilization needs two samples from one compute incarnation | +| Memory usage and limit | Provider observation | Current and periodic history when reported | Point-in-time guest/container memory, not host memory | +| Token input/output | Measured Session Usage | Public current Session and periodic Runtime history | Reported canonical usage; absent usage remains unknown | +| Request count, latency, errors | Completed API responses, stored in minute buckets | Fixed route families, methods, outcomes, and latency bins; collector heartbeat tracks quiet intervals | Transport health, not Turn success | +| Terminal Turn outcome and duration | Persisted `turns` terminal timestamps | Completed, failed, and cancelled Turns | Execution outcome, not HTTP success | +| Model distribution and latency | Native per-attempt evidence is not normalized across harnesses | Unavailable | Do not label configured models, cumulative Usage, or Turns as invoked models | +| Tool calls and failures | Sanitized `tool_call` completion events | Attempts with an after event; missing before leaves duration null unless native duration exists | Does not count Agent tool declarations or model selection | +| Sandbox nodes | Deployment administrator node list | Current online, provider-ready, active, retained and capacity fields | A node being online does not prove a Turn can execute | +| Core process, database, queue, and jobs | Core service metrics projection | Current and bounded process-local history | Service health, not Session-scoped execution evidence | + +Current Runtime history has a 30-second default periodic collection interval and +seven-day retention; the public Web ranges are 1h, 6h, and 24h. Its table is +renamed in place from `runtime_history_samples` to +`observability_runtime_samples`. Its capability route distinguishes periodic +collection from on-read collection. Missing values and empty intervals are not +observed zeroes. + +## Collection boundary and remaining work + +The authenticated operator metrics service runs behind Core. It keeps the pinned +Agents API resources unchanged, aggregates sanitized events server-side, and +exposes a bounded read-only extension through +`packages/agents-client`. Browser input may select a fixed range and resolution; +it must not select tenants, storage labels, arbitrary PromQL, endpoints or keys. + +1. **Ingress (implemented):** count completed HTTP requests by route family, + method and coarse outcome (`success`, `client_error`, `server_error`). Record a + latency histogram after the response completes. Health polling and the metric + read itself are excluded. The bounded asynchronous queue records drops and + write failures; failed batches are not replayed onto the request path. +2. **Execution (implemented):** query terminal Turns from durable state with queue + wait and run duration measured from persisted timestamps. The query is bounded + to a fixed time window and avoids duplicate terminal events. Active Turns + remain a current Core gauge, not a counter inferred from sampled CPU. +3. **Model (pending):** observe actual native model invocation attempts at the harness + boundary, including model identifier, terminal attempt outcome, duration, + reported input/output tokens and usage coverage. A saved Agent's configured + model is not proof of which model ran. Keep bounded model label cardinality. +4. **Tools (partial):** observe `tool_call` before/after events at the common + execution journal boundary. Use a bounded tool category and outcome label; + raw tool names, arguments, output and credentials stay out of metric labels. + Record a completed attempt only after the corresponding event batch has been + persisted. Completion is deduplicated by Turn and call ID. A missing after + event is not counted as a completed attempt. Some harnesses may not emit + these events. +5. **Sandbox (partial):** retain allocation identity and compute generation internally. + Extend the normalized sample only after Docker and microsandbox values have + matching semantics. Candidates are OOM/exit events, disk usage/limit, network + bytes, compute restarts, and allocation/compute startup latency. A missing + provider value remains null; lifecycle state remains Core-owned. +6. **Collector health (partial):** request collection writes a minute heartbeat, + including quiet minutes. Request and tool queues report drops and write + failures. Runtime sampler and model-attempt collector health are not yet wired + into this operator projection; an absent row is unavailable, not zero. + +The operator migration adds `observability_request_minute_buckets`, +`observability_model_attempts`, `observability_tool_attempts`, and +`observability_collector_minute_buckets`. Request and collector buckets retain +30 days; model and tool attempts retain seven days. Model attempts remain empty +until a native per-attempt producer is implemented. The deployment administrator +summary supports only 1h, 6h, and 24h and returns no tenant or raw identifiers. +It is served at `/core/v1/admin/observability` through the same administrator +credential and `AdminClient` transport as Core service metrics. The existing +`/core/v1/admin/core-metrics` projection supplies queue, database, job, and +process measurements; this collection does not duplicate those sources. + +Use bounded histograms for p50/p95 latency and rates from counters over complete +time buckets. Keep operational cardinality to route family, outcome, provider +type and bounded model family. Session, allocation, tenant, tool name and native +identifiers must not become general metrics labels. Drill-down can use authorized +Core resource IDs through existing tenant-scoped reads. Retention and query limits +must be explicit, with history storage/export optional for execution. + +## Dashboard layout and acceptance + +The overview keeps a small current snapshot: Agents, Sessions, active Sandboxes, +reported tokens, measured CPU/memory capacity, and ready nodes. The Observability +tab leads with the four existing Runtime trends, followed by request, Turn, tool, +collector, node, and Core service panels. The model panel states that collection is unavailable. The +request rate, error rate, and range latency require complete collector heartbeat +coverage with no recorded drops or export failures; otherwise they are unavailable. Request route tables show counts only +for covered buckets. Every panel must preserve missing measurements and must not equate +a configured Sandbox with a completed execution. + +Before shipping those new panels, validate counter deduplication under retries, +provider restarts, missing usage, sampler outages and multi-tenant authorization; +verify that instrumentation has no effect on dispatch, lifecycle or settlement. diff --git a/packages/agents-client/src/admin-client.test.ts b/packages/agents-client/src/admin-client.test.ts index d6d512a25..fa009fc3d 100644 --- a/packages/agents-client/src/admin-client.test.ts +++ b/packages/agents-client/src/admin-client.test.ts @@ -35,6 +35,8 @@ const routeCases: Array<[string, string, (client: AdminClient) => Promise client.retrieveStartupConfiguration()], ["GET", "/audit-log", (client) => client.listAuditLog()], ["GET", "/summary", (client) => client.retrieveSummary()], + ["GET", "/observability?range=1h", (client) => client.retrieveObservability("1h")], + ["GET", "/core-metrics?range=1h", (client) => client.retrieveCoreMetrics("1h")], ["GET", "/runtime-observations", (client) => client.listRuntimeObservations()], ["GET", `/projects/${projectId}/agents`, (client) => client.listAgents(projectId)], ["GET", `/projects/${projectId}/agents/a`, (client) => client.retrieveAgent(projectId, "a")], @@ -77,6 +79,13 @@ const routeCases: Array<[string, string, (client: AdminClient) => Promise { + it("reads Sandbox nodes through the same administrator transport without a project credential", async () => { + const fetch = vi.fn().mockImplementation(async () => json({ data: [] })); + await expect(new AdminClient({ fetch }).listSandboxNodes()).resolves.toEqual({ data: [] }); + expect(fetch).toHaveBeenCalledOnce(); + expect(fetch.mock.calls[0]?.[0]).toBe("/core/v1/sandbox/nodes"); + expect(fetch.mock.calls[0]?.[1]).toMatchObject({ method: "GET", credentials: "same-origin", redirect: "error" }); + }); it.each(routeCases)("routes %s %s without credential or public API fallback", async (method, path, call) => { const fetch = vi.fn().mockImplementation(async () => json({ error: { message: "not found", code: "not_found" } }, 404)); const client = new AdminClient({ fetch }); diff --git a/packages/agents-client/src/admin-client.ts b/packages/agents-client/src/admin-client.ts index 4f512b61c..95697d493 100644 --- a/packages/agents-client/src/admin-client.ts +++ b/packages/agents-client/src/admin-client.ts @@ -7,6 +7,9 @@ import { projectExecutionConfiguration } from "./execution-configuration-project import { projectAgentTurn, projectSessionItem, projectHistoryPage, validateHistoryPageOptions } from "./history-projection"; import { projectRuntimeHistory, projectRuntimeHistoryCapabilities } from "./runtime-history-projection"; import { canonicalUuid, isNonnegativeInteger, isRecord } from "./response-projection"; +import type { SandboxNode } from "./sandbox-client"; +import type { OperatorMetricsRange, OperatorMetricsSummary } from "./observability-client"; +import type { CoreMetricsView } from "./core-metrics-types"; import { invalidAdminResponse, projectAdminProject, projectAdminKey, projectIssuedAdminKey, projectAdminPage, projectAdminDeleted, projectResourcePage, projectSavedAgent, projectSkill, projectSkillVersion, projectArtifact, projectCopyResult, projectSummary, projectAdminRuntimePage, projectResourceOwners, projectWriteOperations, projectAdminAudit, @@ -60,7 +63,9 @@ export class AdminClient { if (!idempotencyKey.trim()) throw new TypeError("An Idempotency-Key must not be empty."); headers.set("Idempotency-Key", idempotencyKey); } - const response = await this.#fetch(`${this.#baseUrl}${path}`, { + const base = path.startsWith("/sandbox/") && this.#baseUrl.endsWith("/admin") + ? this.#baseUrl.slice(0, -"/admin".length) : this.#baseUrl; + const response = await this.#fetch(`${base}${path}`, { method, headers, signal: options?.signal, credentials: "same-origin", redirect: "error", ...(body === undefined ? {} : { body: JSON.stringify(body) }), }); @@ -128,6 +133,24 @@ export class AdminClient { }); return projectSummary(await this.#json(path, options)); } + async retrieveObservability(range: OperatorMetricsRange, options?: ReadOptions): Promise { + const value = await this.#json(`/observability?range=${range}`, options); + if (!isRecord(value) || !Array.isArray(value.requests) || !Array.isArray(value.collector) || + !Array.isArray(value.turns) || !Array.isArray(value.tools)) return invalidAdminResponse(); + return value as unknown as OperatorMetricsSummary; + } + async retrieveCoreMetrics(range: OperatorMetricsRange, options?: ReadOptions): Promise { + const value = await this.#json(`/core-metrics?range=${range}`, options); + if (!isRecord(value) || value.object !== "core.metrics" || !isRecord(value.service) || + !isRecord(value.execution) || !isRecord(value.database) || !Array.isArray(value.jobs) || !isRecord(value.process)) return invalidAdminResponse(); + return value as unknown as CoreMetricsView; + } + async listSandboxNodes(options?: ReadOptions): Promise<{ data: SandboxNode[] }> { + if (!this.#baseUrl.endsWith("/admin")) throw new TypeError("Sandbox administration requires a Core admin base URL."); + const value = await this.#json("/sandbox/nodes", options); + if (!isRecord(value) || !Array.isArray(value.data)) return invalidAdminResponse(); + return value as unknown as { data: SandboxNode[] }; + } async listRuntimeObservations(options?: PageOptions) { return projectAdminRuntimePage(await this.#json(pageQuery("/runtime-observations", options), options)); } diff --git a/packages/agents-client/src/core-metrics-types.ts b/packages/agents-client/src/core-metrics-types.ts new file mode 100644 index 000000000..1a3b6b0b6 --- /dev/null +++ b/packages/agents-client/src/core-metrics-types.ts @@ -0,0 +1,27 @@ +/** Bounded deployment metrics returned by the administrator API. */ +export interface CoreMetricsView { + object: "core.metrics"; + range: { start: string; end: string; resolution_seconds: number }; + service: { status: string; revision: string | null; started_at: string | null; execution_owner: boolean | null }; + execution: { + slots_in_use: number | null; + slots_total: number | null; + queued_turns: number | null; + waiting_for_daemon: number | null; + in_progress_turns: number | null; + oldest_queued_seconds: number | null; + connected_daemons: number | null; + interrupted: number | null; + unavailable: number | null; + queue_wait_ms: { p50: number | null; p95: number | null }; + series: Array<{ start: string; queued: number | null; in_progress: number | null; queue_wait_p95_ms: number | null }>; + }; + database: { + ping_ms: { p50: number | null; p95: number | null }; + pool: { in_use: number | null; idle: number | null; max: number | null }; + size_bytes: number | null; + series: Array<{ start: string; ping_p95_ms: number | null; pool_in_use: number | null }>; + }; + jobs: Array<{ id: string; status: string; last_run_at: string | null; processed: number | null; failed: number | null }>; + process: { memory_bytes: number | null; goroutines: number | null }; +} diff --git a/packages/agents-client/src/index.ts b/packages/agents-client/src/index.ts index 5cc357c5f..c883122fc 100644 --- a/packages/agents-client/src/index.ts +++ b/packages/agents-client/src/index.ts @@ -4,5 +4,7 @@ export { createSSEDecoder } from "./sse"; export type { SSEDecoder, SSEMessage } from "./sse"; export type * from "./types"; export * from "./sandbox-client"; +export * from "./observability-client"; +export type { CoreMetricsView } from "./core-metrics-types"; export { AdminClient } from "./admin-client"; export type * from "./admin-types"; diff --git a/packages/agents-client/src/observability-client.ts b/packages/agents-client/src/observability-client.ts new file mode 100644 index 000000000..210024b43 --- /dev/null +++ b/packages/agents-client/src/observability-client.ts @@ -0,0 +1,32 @@ +export type OperatorMetricsRange = "1h" | "6h" | "24h"; + +export interface RequestMetricBucket { + start: string; + route_family: string; + outcome: "success" | "client_error" | "server_error"; + count: number; + latency_sum_ms: number; + latency_bucket_counts: number[]; +} + +export interface CollectorMetricBucket { + start: string; + source: string; + attempted_count: number; + observed_count: number; + unavailable_count: number; + timeout_count: number; + dropped_count: number; + export_failed_count: number; +} + +export interface OperatorMetricsSummary { + generated_at: string; + start: string; + end: string; + step_seconds: number; + requests: RequestMetricBucket[]; + collector: CollectorMetricBucket[]; + turns: Array<{ start: string; status: "completed" | "failed" | "cancelled"; count: number; queue_p95_ms: number | null; execution_p95_ms: number | null }>; + tools: Array<{ start: string; category: string; outcome: "success" | "error" | "cancelled" | "unknown"; count: number; timed_count: number; duration_p95_ms: number | null }>; +} diff --git a/services/agents-api/cmd/server/main.go b/services/agents-api/cmd/server/main.go index c86006bfb..559343008 100644 --- a/services/agents-api/cmd/server/main.go +++ b/services/agents-api/cmd/server/main.go @@ -35,6 +35,7 @@ import ( "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/api" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/coremetrics" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/execution" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtime" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeenrollment" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimehistory" @@ -91,6 +92,9 @@ func run() error { return err } executionStore := store.NewWithCredentialCipherAndOAuthRefresh(pool, credentialKey, oauthClient) + toolMetricsCtx, cancelToolMetrics := context.WithCancel(ctx) + toolMetrics := observability.NewToolRecorder(toolMetricsCtx, executionStore) + defer func() { cancelToolMetrics(); <-toolMetrics.Done() }() metricsSource := &coreMetricsSource{store: executionStore, pool: pool} metrics := coremetrics.New(processStartedAt, buildRevision, metricsSource) auth, err := api.NewDatabaseAuthenticator(executionStore) @@ -181,6 +185,7 @@ func run() error { return err } options = append(options, api.WithProjectAPIKeys(executionStore, keyAdmin), api.WithWriteAudit(executionStore, keyAdmin), api.WithAdminManagement(executionStore)) + options = append(options, api.WithOperatorMetrics(executionStore, keyAdmin)) if history.Reader != nil { historyResolver, resolverErr := historystoreresolver.NewResolver(executionStore) if resolverErr != nil { @@ -204,7 +209,7 @@ func run() error { } if registry != nil { dispatcher := &execution.Dispatcher{Store: executionStore, Registry: registry, - ManagedRuntimes: managed, Options: transientOptions} + ManagedRuntimes: managed, Options: transientOptions, ToolRecorder: toolMetrics} worker, err = execution.StartWorker(ctx, dispatcher) if err != nil { @@ -271,6 +276,17 @@ func run() error { startupManaged = managedNodes.setup.selected.Load() } options = append(options, api.WithStartupConfiguration(coreStartupConfiguration(engine, kinds, registry != nil, modelProviderEndpoints, managedRuntimeProviderKind(startupManaged), startupManaged))) + requestMetricsCtx, cancelRequestMetrics := context.WithCancel(ctx) + requestMetrics := observability.NewRequestRecorder(requestMetricsCtx, executionStore) + defer func() { cancelRequestMetrics(); <-requestMetrics.Done() }() + options = append(options, api.WithRequestMetrics(requestMetrics)) + operatorCleanupCtx, cancelOperatorCleanup := context.WithCancel(ctx) + operatorCleanupDone := make(chan struct{}) + go func() { + defer close(operatorCleanupDone) + runOperatorMetricsCleanup(operatorCleanupCtx, executionStore) + }() + defer func() { cancelOperatorCleanup(); <-operatorCleanupDone }() handler, err := api.NewHandler(executionStore, auth, engine, options...) if err != nil { return err diff --git a/services/agents-api/cmd/server/observability_cleanup.go b/services/agents-api/cmd/server/observability_cleanup.go new file mode 100644 index 000000000..ed835f2fc --- /dev/null +++ b/services/agents-api/cmd/server/observability_cleanup.go @@ -0,0 +1,30 @@ +package main + +import ( + "context" + "time" + + "github.com/MiniMax-AI-Dev/parsar/internal/obs/log" +) + +type operatorMetricsPruner interface { + PruneOperatorMetrics(context.Context, time.Time) error +} + +func runOperatorMetricsCleanup(ctx context.Context, pruner operatorMetricsPruner) { + ticker := time.NewTicker(time.Minute) + defer ticker.Stop() + for { + pruneCtx, cancel := context.WithTimeout(ctx, 3*time.Second) + err := pruner.PruneOperatorMetrics(pruneCtx, time.Now().UTC()) + cancel() + if err != nil && ctx.Err() == nil { + log.Ctx(ctx).Warn("Operator metrics retention cleanup failed") + } + select { + case <-ctx.Done(): + return + case <-ticker.C: + } + } +} diff --git a/services/agents-api/internal/api/handler.go b/services/agents-api/internal/api/handler.go index fc65c568c..f8e24f243 100644 --- a/services/agents-api/internal/api/handler.go +++ b/services/agents-api/internal/api/handler.go @@ -13,6 +13,7 @@ import ( "github.com/MiniMax-AI-Dev/parsar/internal/obs/log" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/execution" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/identity" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" "github.com/go-chi/chi/v5" "github.com/go-chi/chi/v5/middleware" @@ -63,6 +64,8 @@ type Handler struct { subagents SubagentStore runtimeObservations RuntimeObservationService runtimeHistory RuntimeHistoryService + requestMetrics *observability.RequestRecorder + operatorMetrics OperatorMetricsStore startup *v1.CoreStartupConfiguration } @@ -87,6 +90,9 @@ func NewHandler(s ResourceStore, auth *Authenticator, engine string, options ... func (h *Handler) routes() *chi.Mux { router := chi.NewRouter() router.Use(h.responseHeaders, log.HTTPMiddleware, middleware.GetHead) + if h.requestMetrics != nil { + router.Use(h.recordRequestMetrics) + } router.MethodNotAllowed(methodNotAllowed) router.Get("/healthz", func(w http.ResponseWriter, _ *http.Request) { writeJSON(w, http.StatusOK, map[string]string{"status": "ok"}) @@ -102,11 +108,12 @@ func (h *Handler) routes() *chi.Mux { r.Delete("/v1/files/{file_id}", h.deleteSourceFile) }) h.registerSandboxManagerRoutes(router) - if h.deploymentAuth != nil && h.projectKeys != nil { + if h.deploymentAuth != nil { router.Route("/core/v1/admin", func(r chi.Router) { r.Use(h.deploymentAuth.authenticate) h.registerProjectAPIKeyRoutes(r) h.registerAdminResourceRoutes(r) + h.registerOperatorMetricsRoutes(r) }) } h.registerEnvironmentExecutorRoutes(router) diff --git a/services/agents-api/internal/api/observability_read.go b/services/agents-api/internal/api/observability_read.go new file mode 100644 index 000000000..0ef55e34f --- /dev/null +++ b/services/agents-api/internal/api/observability_read.go @@ -0,0 +1,85 @@ +package api + +import ( + "context" + "net/http" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" + "github.com/go-chi/chi/v5" +) + +type OperatorMetricsStore interface { + ReadOperatorMetrics(context.Context, time.Time, time.Time, time.Duration) (store.OperatorMetrics, error) +} + +type OperatorMetricsResponse struct { + GeneratedAt time.Time `json:"generated_at"` + Start time.Time `json:"start"` + End time.Time `json:"end"` + StepSeconds int64 `json:"step_seconds"` + Requests []store.RequestMetricBucket `json:"requests"` + Collector []store.CollectorMetricBucket `json:"collector"` + Turns []store.TurnMetricBucket `json:"turns"` + Tools []store.ToolMetricBucket `json:"tools"` +} + +func WithOperatorMetrics(s OperatorMetricsStore, auth *DeploymentAuthenticator) Option { + return func(h *Handler) { + h.operatorMetrics = s + if auth != nil { + h.deploymentAuth = auth + } + } +} + +func (h *Handler) registerOperatorMetricsRoutes(r chi.Router) { + if h.operatorMetrics == nil { + return + } + r.Get("/observability", h.getOperatorMetrics) + r.Head("/observability", methodNotAllowed) +} + +// @Summary Retrieve bounded operator observability metrics +// @Description Deployment administrator only. Returns fixed-range aggregate requests and collector coverage without paths, payloads or project identifiers. +// @Tags Observability +// @Produce json +// @Security DeploymentAdminAuth +// @Param range query string false "1h, 6h, or 24h (default 1h)" +// @Success 200 {object} api.OperatorMetricsResponse +// @Failure 400,401,500,503 {object} v1.ErrorResponse +// @Router /core/v1/admin/observability [get] +func (h *Handler) getOperatorMetrics(w http.ResponseWriter, r *http.Request) { + values := r.URL.Query() + for key, entries := range values { + if key != "range" || len(entries) != 1 { + writeStoreError(w, r, store.ErrInvalidInput) + return + } + } + span, step := time.Hour, time.Minute + switch values.Get("range") { + case "", "1h": + case "6h": + span, step = 6*time.Hour, 5*time.Minute + case "24h": + span, step = 24*time.Hour, 15*time.Minute + default: + writeStoreError(w, r, store.ErrInvalidInput) + return + } + end := time.Now().UTC().Truncate(step) + start := end.Add(-span) + ctx, cancel := context.WithTimeout(r.Context(), 5*time.Second) + defer cancel() + metrics, err := h.operatorMetrics.ReadOperatorMetrics(ctx, start, end, step) + if err != nil { + writeStoreError(w, r, err) + return + } + writeJSON(w, http.StatusOK, OperatorMetricsResponse{ + GeneratedAt: time.Now().UTC(), Start: start, End: end, StepSeconds: int64(step / time.Second), + Requests: metrics.Requests, Collector: metrics.Collector, Turns: metrics.Turns, Tools: metrics.Tools, + }) +} diff --git a/services/agents-api/internal/api/observability_read_test.go b/services/agents-api/internal/api/observability_read_test.go new file mode 100644 index 000000000..5dab16750 --- /dev/null +++ b/services/agents-api/internal/api/observability_read_test.go @@ -0,0 +1,62 @@ +package api + +import ( + "context" + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" + + "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/device" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" +) + +type operatorMetricsStub struct{ calls int } + +func (s *operatorMetricsStub) ReadOperatorMetrics(_ context.Context, start, end time.Time, step time.Duration) (store.OperatorMetrics, error) { + s.calls++ + if end.Sub(start) != time.Hour || step != time.Minute { + return store.OperatorMetrics{}, store.ErrInvalidInput + } + return store.OperatorMetrics{Requests: []store.RequestMetricBucket{}, Collector: []store.CollectorMetricBucket{}, Turns: []store.TurnMetricBucket{}, Tools: []store.ToolMetricBucket{}}, nil +} + +func TestOperatorMetricsRequireAdministratorAndBoundedRange(t *testing.T) { + project, err := NewAuthenticator([]APIKey{callerBinding()}) + if err != nil { + t.Fatal(err) + } + admin, err := NewDeploymentAuthenticator([]string{device.HashCredential("administrator")}) + if err != nil { + t.Fatal(err) + } + metrics := &operatorMetricsStub{} + h, err := NewHandler(&recordingStore{}, project, "codex", WithOperatorMetrics(metrics, admin)) + if err != nil { + t.Fatal(err) + } + request := func(path, key string) *httptest.ResponseRecorder { + req := httptest.NewRequest(http.MethodGet, path, nil) + req.Header.Set("Authorization", "Bearer "+key) + response := httptest.NewRecorder() + h.ServeHTTP(response, req) + return response + } + for _, key := range []string{"caller", ""} { + response := request("/core/v1/admin/observability?range=1h", key) + if response.Code != http.StatusUnauthorized || metrics.calls != 0 { + t.Fatal("project or missing key reached operator metrics", response.Code, metrics.calls) + } + } + for _, path := range []string{"/core/v1/admin/observability?range=30d", "/core/v1/admin/observability?range=1h&range=24h", "/core/v1/admin/observability?tenant_id=x"} { + response := request(path, "administrator") + if response.Code != http.StatusBadRequest || metrics.calls != 0 { + t.Fatal("unbounded metrics query admitted", response.Code, metrics.calls) + } + } + response := request("/core/v1/admin/observability?range=1h", "administrator") + if response.Code != http.StatusOK || metrics.calls != 1 || !strings.Contains(response.Body.String(), `"requests":[]`) { + t.Fatal("administrator metrics read failed", response.Code, metrics.calls, response.Body.String()) + } +} diff --git a/services/agents-api/internal/api/observability_requests.go b/services/agents-api/internal/api/observability_requests.go new file mode 100644 index 000000000..5e5366735 --- /dev/null +++ b/services/agents-api/internal/api/observability_requests.go @@ -0,0 +1,76 @@ +package api + +import ( + "net/http" + "strings" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" + "github.com/go-chi/chi/v5" + "github.com/go-chi/chi/v5/middleware" +) + +func WithRequestMetrics(recorder *observability.RequestRecorder) Option { + return func(h *Handler) { h.requestMetrics = recorder } +} + +func (h *Handler) recordRequestMetrics(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.URL.Path == "/healthz" || r.URL.Path == "/core/v1/admin/observability" { + next.ServeHTTP(w, r) + return + } + start := time.Now() + wrapped := middleware.NewWrapResponseWriter(w, r.ProtoMajor) + next.ServeHTTP(wrapped, r) + status := wrapped.Status() + if status == 0 { + status = http.StatusOK + } + outcome := "success" + if status >= 500 { + outcome = "server_error" + } else if status >= 400 { + outcome = "client_error" + } + method := r.Method + switch method { + case http.MethodGet, http.MethodPost, http.MethodPut, http.MethodPatch, http.MethodDelete, http.MethodHead: + default: + method = "OTHER" + } + pattern := "" + if route := chi.RouteContext(r.Context()); route != nil { + pattern = route.RoutePattern() + } + h.requestMetrics.Record(observability.Request{ + CompletedAt: time.Now().UTC(), RouteFamily: requestRouteFamily(pattern), + Method: method, Outcome: outcome, Latency: time.Since(start), + }) + }) +} + +func requestRouteFamily(pattern string) string { + switch { + case strings.HasPrefix(pattern, "/core/v1/sandbox"): + return "sandbox_admin" + case strings.HasPrefix(pattern, "/core/v1"): + return "core_admin" + case strings.HasPrefix(pattern, "/v1/agents/sessions"): + return "sessions" + case strings.HasPrefix(pattern, "/v1/agents/runtime"): + return "runtime" + case strings.HasPrefix(pattern, "/v1/agents/environments"): + return "environments" + case strings.HasPrefix(pattern, "/v1/agents"): + return "agents" + case strings.HasPrefix(pattern, "/v1/files"): + return "files" + case strings.HasPrefix(pattern, "/v1/vaults"): + return "vaults" + case strings.HasPrefix(pattern, "/v1/skills"): + return "skills" + default: + return "other" + } +} diff --git a/services/agents-api/internal/db/queries/runtime_history.sql b/services/agents-api/internal/db/queries/runtime_history.sql index d0c019024..6a20f0a00 100644 --- a/services/agents-api/internal/db/queries/runtime_history.sql +++ b/services/agents-api/internal/db/queries/runtime_history.sql @@ -1,5 +1,5 @@ -- name: InsertRuntimeHistorySample :exec -INSERT INTO runtime_history_samples ( +INSERT INTO observability_runtime_samples ( tenant_id, session_id, environment_id, resolved_at_ns, allocation_id, provider_type, status, observed_at_ns, started_at_ns, cpu_usage_seconds, cpu_capacity_cores, memory_usage_bytes, memory_limit_bytes, input_tokens, output_tokens @@ -12,7 +12,7 @@ WHERE s.tenant_id = sqlc.arg(tenant_id) AND s.id = sqlc.arg(session_id) AND e.id ON CONFLICT (tenant_id, session_id, environment_id, resolved_at_ns) DO NOTHING; -- name: ListRuntimeHistorySamples :many -SELECT * FROM runtime_history_samples +SELECT * FROM observability_runtime_samples WHERE tenant_id = sqlc.arg(tenant_id) AND session_id = sqlc.arg(session_id) AND environment_id = sqlc.arg(environment_id) AND resolved_at_ns >= sqlc.arg(start_ns) AND resolved_at_ns < sqlc.arg(end_ns) ORDER BY resolved_at_ns @@ -20,10 +20,10 @@ LIMIT sqlc.arg(row_limit); -- name: PruneRuntimeHistorySamples :execrows WITH expired AS ( - SELECT p.tenant_id, p.session_id, p.environment_id, p.resolved_at_ns FROM runtime_history_samples p + SELECT p.tenant_id, p.session_id, p.environment_id, p.resolved_at_ns FROM observability_runtime_samples p WHERE p.resolved_at_ns < sqlc.arg(before_ns) ORDER BY p.resolved_at_ns LIMIT 256 FOR UPDATE SKIP LOCKED ) -DELETE FROM runtime_history_samples h USING expired e +DELETE FROM observability_runtime_samples h USING expired e WHERE h.tenant_id = e.tenant_id AND h.session_id = e.session_id AND h.environment_id = e.environment_id AND h.resolved_at_ns = e.resolved_at_ns; diff --git a/services/agents-api/internal/db/sqlc/models.go b/services/agents-api/internal/db/sqlc/models.go index 41380edb8..a6287f37f 100644 --- a/services/agents-api/internal/db/sqlc/models.go +++ b/services/agents-api/internal/db/sqlc/models.go @@ -167,6 +167,73 @@ type InitialEnvironmentFile struct { Contents []byte `json:"contents"` } +type ObservabilityCollectorMinuteBucket struct { + BucketStart pgtype.Timestamptz `json:"bucket_start"` + Source string `json:"source"` + AttemptedCount int64 `json:"attempted_count"` + ObservedCount int64 `json:"observed_count"` + UnavailableCount int64 `json:"unavailable_count"` + TimeoutCount int64 `json:"timeout_count"` + DroppedCount int64 `json:"dropped_count"` + ExportFailedCount int64 `json:"export_failed_count"` +} + +type ObservabilityModelAttempt struct { + ID pgtype.UUID `json:"id"` + TenantID pgtype.UUID `json:"tenant_id"` + SessionID pgtype.UUID `json:"session_id"` + TurnID pgtype.UUID `json:"turn_id"` + StartedAt pgtype.Timestamptz `json:"started_at"` + FinishedAt pgtype.Timestamptz `json:"finished_at"` + ModelFamily string `json:"model_family"` + ProviderType string `json:"provider_type"` + Outcome string `json:"outcome"` + DurationMs int64 `json:"duration_ms"` + InputTokens pgtype.Int8 `json:"input_tokens"` + OutputTokens pgtype.Int8 `json:"output_tokens"` + UsageStatus string `json:"usage_status"` +} + +type ObservabilityRequestMinuteBucket struct { + BucketStart pgtype.Timestamptz `json:"bucket_start"` + RouteFamily string `json:"route_family"` + Method string `json:"method"` + Outcome string `json:"outcome"` + RequestCount int64 `json:"request_count"` + LatencySumMs int64 `json:"latency_sum_ms"` + LatencyBucketCounts []int64 `json:"latency_bucket_counts"` +} + +type ObservabilityRuntimeSample struct { + TenantID pgtype.UUID `json:"tenant_id"` + SessionID pgtype.UUID `json:"session_id"` + EnvironmentID pgtype.UUID `json:"environment_id"` + ResolvedAtNs int64 `json:"resolved_at_ns"` + AllocationID pgtype.UUID `json:"allocation_id"` + ProviderType string `json:"provider_type"` + Status string `json:"status"` + ObservedAtNs pgtype.Int8 `json:"observed_at_ns"` + StartedAtNs pgtype.Int8 `json:"started_at_ns"` + CpuUsageSeconds pgtype.Float8 `json:"cpu_usage_seconds"` + CpuCapacityCores pgtype.Float8 `json:"cpu_capacity_cores"` + MemoryUsageBytes pgtype.Int8 `json:"memory_usage_bytes"` + MemoryLimitBytes pgtype.Int8 `json:"memory_limit_bytes"` + InputTokens pgtype.Int8 `json:"input_tokens"` + OutputTokens pgtype.Int8 `json:"output_tokens"` +} + +type ObservabilityToolAttempt struct { + ID pgtype.UUID `json:"id"` + TenantID pgtype.UUID `json:"tenant_id"` + SessionID pgtype.UUID `json:"session_id"` + TurnID pgtype.UUID `json:"turn_id"` + StartedAt pgtype.Timestamptz `json:"started_at"` + FinishedAt pgtype.Timestamptz `json:"finished_at"` + ToolCategory string `json:"tool_category"` + Outcome string `json:"outcome"` + DurationMs pgtype.Int8 `json:"duration_ms"` +} + type Project struct { ID pgtype.UUID `json:"id"` Name string `json:"name"` @@ -249,24 +316,6 @@ type RuntimeDeviceAuthority struct { CredentialHash string `json:"credential_hash"` } -type RuntimeHistorySample struct { - TenantID pgtype.UUID `json:"tenant_id"` - SessionID pgtype.UUID `json:"session_id"` - EnvironmentID pgtype.UUID `json:"environment_id"` - ResolvedAtNs int64 `json:"resolved_at_ns"` - AllocationID pgtype.UUID `json:"allocation_id"` - ProviderType string `json:"provider_type"` - Status string `json:"status"` - ObservedAtNs pgtype.Int8 `json:"observed_at_ns"` - StartedAtNs pgtype.Int8 `json:"started_at_ns"` - CpuUsageSeconds pgtype.Float8 `json:"cpu_usage_seconds"` - CpuCapacityCores pgtype.Float8 `json:"cpu_capacity_cores"` - MemoryUsageBytes pgtype.Int8 `json:"memory_usage_bytes"` - MemoryLimitBytes pgtype.Int8 `json:"memory_limit_bytes"` - InputTokens pgtype.Int8 `json:"input_tokens"` - OutputTokens pgtype.Int8 `json:"output_tokens"` -} - type RuntimeNode struct { ID pgtype.UUID `json:"id"` InstallationID pgtype.UUID `json:"installation_id"` diff --git a/services/agents-api/internal/db/sqlc/runtime_history.sql.go b/services/agents-api/internal/db/sqlc/runtime_history.sql.go index 1c6a84eac..24567a81c 100644 --- a/services/agents-api/internal/db/sqlc/runtime_history.sql.go +++ b/services/agents-api/internal/db/sqlc/runtime_history.sql.go @@ -12,7 +12,7 @@ import ( ) const insertRuntimeHistorySample = `-- name: InsertRuntimeHistorySample :exec -INSERT INTO runtime_history_samples ( +INSERT INTO observability_runtime_samples ( tenant_id, session_id, environment_id, resolved_at_ns, allocation_id, provider_type, status, observed_at_ns, started_at_ns, cpu_usage_seconds, cpu_capacity_cores, memory_usage_bytes, memory_limit_bytes, input_tokens, output_tokens @@ -65,7 +65,7 @@ func (q *Queries) InsertRuntimeHistorySample(ctx context.Context, arg InsertRunt } const listRuntimeHistorySamples = `-- name: ListRuntimeHistorySamples :many -SELECT tenant_id, session_id, environment_id, resolved_at_ns, allocation_id, provider_type, status, observed_at_ns, started_at_ns, cpu_usage_seconds, cpu_capacity_cores, memory_usage_bytes, memory_limit_bytes, input_tokens, output_tokens FROM runtime_history_samples +SELECT tenant_id, session_id, environment_id, resolved_at_ns, allocation_id, provider_type, status, observed_at_ns, started_at_ns, cpu_usage_seconds, cpu_capacity_cores, memory_usage_bytes, memory_limit_bytes, input_tokens, output_tokens FROM observability_runtime_samples WHERE tenant_id = $1 AND session_id = $2 AND environment_id = $3 AND resolved_at_ns >= $4 AND resolved_at_ns < $5 ORDER BY resolved_at_ns @@ -81,7 +81,7 @@ type ListRuntimeHistorySamplesParams struct { RowLimit int32 `json:"row_limit"` } -func (q *Queries) ListRuntimeHistorySamples(ctx context.Context, arg ListRuntimeHistorySamplesParams) ([]RuntimeHistorySample, error) { +func (q *Queries) ListRuntimeHistorySamples(ctx context.Context, arg ListRuntimeHistorySamplesParams) ([]ObservabilityRuntimeSample, error) { rows, err := q.db.Query(ctx, listRuntimeHistorySamples, arg.TenantID, arg.SessionID, @@ -94,9 +94,9 @@ func (q *Queries) ListRuntimeHistorySamples(ctx context.Context, arg ListRuntime return nil, err } defer rows.Close() - items := []RuntimeHistorySample{} + items := []ObservabilityRuntimeSample{} for rows.Next() { - var i RuntimeHistorySample + var i ObservabilityRuntimeSample if err := rows.Scan( &i.TenantID, &i.SessionID, @@ -126,11 +126,11 @@ func (q *Queries) ListRuntimeHistorySamples(ctx context.Context, arg ListRuntime const pruneRuntimeHistorySamples = `-- name: PruneRuntimeHistorySamples :execrows WITH expired AS ( - SELECT p.tenant_id, p.session_id, p.environment_id, p.resolved_at_ns FROM runtime_history_samples p + SELECT p.tenant_id, p.session_id, p.environment_id, p.resolved_at_ns FROM observability_runtime_samples p WHERE p.resolved_at_ns < $1 ORDER BY p.resolved_at_ns LIMIT 256 FOR UPDATE SKIP LOCKED ) -DELETE FROM runtime_history_samples h USING expired e +DELETE FROM observability_runtime_samples h USING expired e WHERE h.tenant_id = e.tenant_id AND h.session_id = e.session_id AND h.environment_id = e.environment_id AND h.resolved_at_ns = e.resolved_at_ns ` diff --git a/services/agents-api/internal/execution/delivery.go b/services/agents-api/internal/execution/delivery.go index 52b6d59dd..9e97f14c3 100644 --- a/services/agents-api/internal/execution/delivery.go +++ b/services/agents-api/internal/execution/delivery.go @@ -65,7 +65,7 @@ func (d *Dispatcher) deliver(ctx context.Context, tenantID, sessionID string, pe } }() journal := &journal{store: d.Store, tenant: tenantID, session: sessionID, turn: request.RunID, next: 1, - observeSubagents: request.ObserveSubagentIdentities} + observeSubagents: request.ObserveSubagentIdentities, toolRecorder: d.ToolRecorder} defer func() { finishCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second) defer cancel() diff --git a/services/agents-api/internal/execution/dispatcher.go b/services/agents-api/internal/execution/dispatcher.go index 3d9c98e03..4370da884 100644 --- a/services/agents-api/internal/execution/dispatcher.go +++ b/services/agents-api/internal/execution/dispatcher.go @@ -11,6 +11,7 @@ import ( v1 "github.com/MiniMax-AI-Dev/parsar/contracts/agents-api/v1" "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/gateway" "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/proto" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" ) @@ -36,6 +37,10 @@ type Dispatcher struct { Options func(context.Context, store.Session) (map[string]any, error) // ManagedRuntimes is optional internal provisioning; it does not admit hosted API requests. ManagedRuntimes *RuntimeProvider + // ToolRecorder receives sanitized terminal observations without affecting execution. + ToolRecorder interface { + Record(observability.ToolAttempt) + } } type Result struct { diff --git a/services/agents-api/internal/execution/journal.go b/services/agents-api/internal/execution/journal.go index ae22830de..0835a42c2 100644 --- a/services/agents-api/internal/execution/journal.go +++ b/services/agents-api/internal/execution/journal.go @@ -6,7 +6,9 @@ import ( "time" "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/proto" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" + "github.com/google/uuid" ) type journal struct { @@ -14,9 +16,14 @@ type journal struct { tenant, session, turn string next int32 batch []store.ExecutionEvent + batchTimes []time.Time bytes int pendingCount int observeSubagents bool + toolRecorder interface { + Record(observability.ToolAttempt) + } + toolStarts map[string]time.Time } type eventWriter interface { @@ -67,10 +74,66 @@ func (j *journal) enqueue(env proto.Envelope) error { return store.ErrEventLimit } j.batch = append(j.batch, store.ExecutionEvent{Kind: env.Type, Payload: env.Payload}) + j.batchTimes = append(j.batchTimes, time.Now().UTC()) j.bytes += len(env.Payload) return nil } +func (j *journal) observeToolAttempt(raw json.RawMessage, observedAt time.Time) { + if j.toolRecorder == nil { + return + } + var call proto.ToolCallPayload + if json.Unmarshal(raw, &call) != nil || call.ID == "" { + return + } + if call.Stage == "before" { + if j.toolStarts == nil { + j.toolStarts = make(map[string]time.Time) + } + if _, exists := j.toolStarts[call.ID]; !exists { + j.toolStarts[call.ID] = observedAt + } + return + } + if call.Stage != "after" { + return + } + finished := observedAt + var started *time.Time + if value, exists := j.toolStarts[call.ID]; exists { + started = &value + delete(j.toolStarts, call.ID) + } + category, outcome := "other", "unknown" + var duration *int64 + if call.Observation != nil { + switch call.Observation.Kind { + case "command", "mcp", "function", "web_search", "file": + category = call.Observation.Kind + } + switch call.Observation.Status { + case "completed": + outcome = "success" + case "failed": + outcome = "error" + } + if call.Observation.DurationMS != nil && *call.Observation.DurationMS >= 0 { + value := *call.Observation.DurationMS + duration = &value + } + } + if duration == nil && started != nil { + value := finished.Sub(*started).Milliseconds() + duration = &value + } + j.toolRecorder.Record(observability.ToolAttempt{ + ID: uuid.NewSHA1(uuid.NameSpaceOID, []byte(j.turn+":"+call.ID)).String(), + TenantID: j.tenant, SessionID: j.session, TurnID: j.turn, + StartedAt: started, FinishedAt: finished, Category: category, Outcome: outcome, DurationMS: duration, + }) +} + func (j *journal) flush(ctx context.Context) error { if len(j.batch) == 0 { return nil @@ -96,12 +159,19 @@ func (j *journal) flush(ctx context.Context) error { if err := j.store.AppendTurnEvents(ctx, j.tenant, j.session, j.turn, j.next, j.batch[:count]); err != nil { return err } + for index, event := range j.batch[:count] { + if event.Kind == proto.TypeToolCall { + j.observeToolAttempt(event.Payload, j.batchTimes[index]) + } + } j.pendingCount = 0 j.next += int32(count) j.bytes -= size j.batch = j.batch[count:] + j.batchTimes = j.batchTimes[count:] } j.batch = nil + j.batchTimes = nil return nil } diff --git a/services/agents-api/internal/execution/observability_tools_test.go b/services/agents-api/internal/execution/observability_tools_test.go new file mode 100644 index 000000000..2a4cabf28 --- /dev/null +++ b/services/agents-api/internal/execution/observability_tools_test.go @@ -0,0 +1,60 @@ +package execution + +import ( + "context" + "errors" + "testing" + + "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/proto" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" +) + +type toolCapture struct{ attempts []observability.ToolAttempt } + +func (c *toolCapture) Record(value observability.ToolAttempt) { c.attempts = append(c.attempts, value) } + +type toolEventWriter struct{ err error } + +func (w *toolEventWriter) AppendTurnEvents(_ context.Context, _, _, _ string, _ int32, _ []store.ExecutionEvent) error { + return w.err +} + +func TestToolAttemptProjectsOnlyTerminalSanitizedEvidence(t *testing.T) { + capture := &toolCapture{} + writer := &toolEventWriter{} + j := &journal{store: writer, tenant: "tenant", session: "session", turn: "turn", toolRecorder: capture} + before, err := proto.NewEnvelope(proto.TypeToolCall, "turn", proto.ToolCallPayload{ + ID: "native-call", Stage: "before", Name: "secret-tool-name", + Observation: &proto.ToolObservation{Kind: "command", Status: "in_progress", Command: "secret command"}, + }) + if err != nil || j.enqueue(before) != nil || len(capture.attempts) != 0 { + t.Fatalf("tool start should not count as terminal: %v", err) + } + after, err := proto.NewEnvelope(proto.TypeToolCall, "turn", proto.ToolCallPayload{ + ID: "native-call", Stage: "after", Name: "secret-tool-name", + Observation: &proto.ToolObservation{Kind: "command", Status: "failed", Command: "secret command"}, + }) + if err != nil || j.enqueue(after) != nil || len(capture.attempts) != 0 { + t.Fatalf("tool attempt must wait for durable event append: %v", err) + } + writer.err = errors.New("event persistence failed") + if j.flush(context.Background()) == nil || len(capture.attempts) != 0 { + t.Fatal("failed event append produced a tool metric") + } + writer.err = nil + if err := j.flush(context.Background()); err != nil || len(capture.attempts) != 1 { + t.Fatalf("durable terminal tool observation missing: %v", err) + } + got := capture.attempts[0] + if got.Category != "command" || got.Outcome != "error" || got.StartedAt == nil || got.DurationMS == nil || + got.ID == "native-call" || got.ID == "secret-tool-name" || got.TurnID != "turn" { + t.Fatalf("tool attempt lost its bounded projection: %+v", got) + } + if err := j.enqueue(after); err != nil { + t.Fatal(err) + } + if err := j.flush(context.Background()); err != nil || capture.attempts[1].ID != got.ID { + t.Fatalf("retry did not retain deterministic attempt identity: %v", err) + } +} diff --git a/services/agents-api/internal/observability/requests.go b/services/agents-api/internal/observability/requests.go new file mode 100644 index 000000000..d871906de --- /dev/null +++ b/services/agents-api/internal/observability/requests.go @@ -0,0 +1,134 @@ +// Package observability collects bounded, side-channel operator metrics. +package observability + +import ( + "context" + "sync/atomic" + "time" +) + +// Request contains only bounded labels; it never carries a URL, principal, or payload. +type Request struct { + CompletedAt time.Time + RouteFamily string + Method string + Outcome string + Latency time.Duration +} + +type RequestBucket struct { + Start time.Time + RouteFamily string + Method string + Outcome string + Count int64 + LatencySumMS int64 + LatencyCounts [11]int64 +} + +type RequestSink interface { + WriteRequestBuckets(context.Context, []RequestBucket, int64, int64) error +} + +// RequestRecorder batches metrics away from the HTTP response path. +type RequestRecorder struct { + queue chan Request + sink RequestSink + dropped atomic.Int64 + failed atomic.Int64 + done chan struct{} +} + +func NewRequestRecorder(ctx context.Context, sink RequestSink) *RequestRecorder { + r := &RequestRecorder{queue: make(chan Request, 2048), sink: sink, done: make(chan struct{})} + go r.run(ctx) + return r +} + +func (r *RequestRecorder) Record(value Request) { + select { + case r.queue <- value: + default: + r.dropped.Add(1) + } +} + +func (r *RequestRecorder) Done() <-chan struct{} { return r.done } + +func (r *RequestRecorder) run(ctx context.Context) { + defer close(r.done) + ticker := time.NewTicker(time.Second) + defer ticker.Stop() + pending := make(map[RequestBucket]*RequestBucket) + var lastHeartbeat time.Time + flush := func() { + buckets := make([]RequestBucket, 0, len(pending)) + var pendingCount int64 + for _, value := range pending { + buckets = append(buckets, *value) + pendingCount += value.Count + } + dropped := r.dropped.Swap(0) + failed := r.failed.Swap(0) + minute := time.Now().UTC().Truncate(time.Minute) + if len(buckets) == 0 && dropped == 0 && failed == 0 && minute.Equal(lastHeartbeat) { + return + } + writeCtx, cancel := context.WithTimeout(context.Background(), 2*time.Second) + err := r.sink.WriteRequestBuckets(writeCtx, buckets, dropped, failed) + cancel() + if err != nil { + r.failed.Add(1) + r.dropped.Add(dropped + pendingCount) + } else { + lastHeartbeat = minute + } + clear(pending) + } + add := func(value Request) { + if value.CompletedAt.IsZero() { + return + } + start := value.CompletedAt.UTC().Truncate(time.Minute) + key := RequestBucket{Start: start, RouteFamily: value.RouteFamily, Method: value.Method, Outcome: value.Outcome} + bucket := pending[key] + if bucket == nil { + bucket = &key + pending[key] = bucket + } + bucket.Count++ + ms := value.Latency.Milliseconds() + if ms < 0 { + ms = 0 + } + bucket.LatencySumMS += ms + bucket.LatencyCounts[latencyIndex(ms)]++ + } + for { + select { + case value := <-r.queue: + add(value) + case <-ticker.C: + flush() + case <-ctx.Done(): + for { + select { + case value := <-r.queue: + add(value) + default: + flush() + return + } + } + } + } +} + +func latencyIndex(ms int64) int { + for i, bound := range [...]int64{10, 25, 50, 100, 250, 500, 1000, 2500, 5000, 10000} { + if ms <= bound { + return i + } + } + return 10 +} diff --git a/services/agents-api/internal/observability/requests_test.go b/services/agents-api/internal/observability/requests_test.go new file mode 100644 index 000000000..c6af0950f --- /dev/null +++ b/services/agents-api/internal/observability/requests_test.go @@ -0,0 +1,40 @@ +package observability + +import ( + "context" + "testing" + "time" +) + +type capturedRequests struct { + buckets []RequestBucket + dropped int64 +} + +func (s *capturedRequests) WriteRequestBuckets(_ context.Context, buckets []RequestBucket, dropped, _ int64) error { + s.buckets = append(s.buckets, buckets...) + s.dropped += dropped + return nil +} + +func TestRequestRecorderAggregatesBoundedLatencyBuckets(t *testing.T) { + ctx, cancel := context.WithCancel(context.Background()) + sink := &capturedRequests{} + recorder := NewRequestRecorder(ctx, sink) + at := time.Date(2026, 9, 25, 10, 20, 30, 0, time.UTC) + recorder.Record(Request{CompletedAt: at, RouteFamily: "sessions", Method: "GET", Outcome: "success", Latency: 50 * time.Millisecond}) + recorder.Record(Request{CompletedAt: at.Add(3 * time.Second), RouteFamily: "sessions", Method: "GET", Outcome: "success", Latency: 11 * time.Second}) + cancel() + select { + case <-recorder.Done(): + case <-time.After(3 * time.Second): + t.Fatal("request recorder did not drain on shutdown") + } + if len(sink.buckets) != 1 { + t.Fatalf("expected one aggregate bucket, got %d", len(sink.buckets)) + } + bucket := sink.buckets[0] + if bucket.Count != 2 || bucket.Start != at.Truncate(time.Minute) || bucket.LatencySumMS != 11050 || bucket.LatencyCounts[2] != 1 || bucket.LatencyCounts[10] != 1 || sink.dropped != 0 { + t.Fatalf("unexpected request aggregate: %+v, dropped=%d", bucket, sink.dropped) + } +} diff --git a/services/agents-api/internal/observability/tools.go b/services/agents-api/internal/observability/tools.go new file mode 100644 index 000000000..ab4d7a34f --- /dev/null +++ b/services/agents-api/internal/observability/tools.go @@ -0,0 +1,90 @@ +package observability + +import ( + "context" + "sync/atomic" + "time" +) + +// ToolAttempt is a sanitized terminal execution observation. +type ToolAttempt struct { + ID string + TenantID string + SessionID string + TurnID string + StartedAt *time.Time + FinishedAt time.Time + Category string + Outcome string + DurationMS *int64 +} + +type ToolSink interface { + WriteToolAttempts(context.Context, []ToolAttempt, int64, int64) error +} + +type ToolRecorder struct { + queue chan ToolAttempt + sink ToolSink + dropped atomic.Int64 + failed atomic.Int64 + done chan struct{} +} + +func NewToolRecorder(ctx context.Context, sink ToolSink) *ToolRecorder { + r := &ToolRecorder{queue: make(chan ToolAttempt, 1024), sink: sink, done: make(chan struct{})} + go r.run(ctx) + return r +} + +func (r *ToolRecorder) Record(value ToolAttempt) { + select { + case r.queue <- value: + default: + r.dropped.Add(1) + } +} + +func (r *ToolRecorder) Done() <-chan struct{} { return r.done } + +func (r *ToolRecorder) run(ctx context.Context) { + defer close(r.done) + ticker := time.NewTicker(time.Second) + defer ticker.Stop() + pending := make([]ToolAttempt, 0, 128) + flush := func() { + dropped, failed := r.dropped.Swap(0), r.failed.Swap(0) + if len(pending) == 0 && dropped == 0 && failed == 0 { + return + } + writeCtx, cancel := context.WithTimeout(context.Background(), 2*time.Second) + err := r.sink.WriteToolAttempts(writeCtx, pending, dropped, failed) + cancel() + if err != nil { + r.failed.Add(1) + r.dropped.Add(dropped + int64(len(pending))) + } + pending = pending[:0] + } + for { + select { + case value := <-r.queue: + pending = append(pending, value) + if len(pending) >= 128 { + flush() + } + case <-ticker.C: + flush() + case <-ctx.Done(): + for { + select { + case value := <-r.queue: + pending = append(pending, value) + default: + flush() + return + } + } + } + } +} diff --git a/services/agents-api/internal/store/observability_read.go b/services/agents-api/internal/store/observability_read.go new file mode 100644 index 000000000..211a0835d --- /dev/null +++ b/services/agents-api/internal/store/observability_read.go @@ -0,0 +1,164 @@ +package store + +import ( + "context" + "time" +) + +type RequestMetricBucket struct { + Start time.Time `json:"start"` + RouteFamily string `json:"route_family"` + Outcome string `json:"outcome"` + Count int64 `json:"count"` + LatencySumMS int64 `json:"latency_sum_ms"` + LatencyCounts []int64 `json:"latency_bucket_counts"` +} + +type CollectorMetricBucket struct { + Start time.Time `json:"start"` + Source string `json:"source"` + AttemptedCount int64 `json:"attempted_count"` + ObservedCount int64 `json:"observed_count"` + UnavailableCount int64 `json:"unavailable_count"` + TimeoutCount int64 `json:"timeout_count"` + DroppedCount int64 `json:"dropped_count"` + ExportFailedCount int64 `json:"export_failed_count"` +} + +type OperatorMetrics struct { + Requests []RequestMetricBucket `json:"requests"` + Collector []CollectorMetricBucket `json:"collector"` + Turns []TurnMetricBucket `json:"turns"` + Tools []ToolMetricBucket `json:"tools"` +} + +type ToolMetricBucket struct { + Start time.Time `json:"start"` + Category string `json:"category"` + Outcome string `json:"outcome"` + Count int64 `json:"count"` + TimedCount int64 `json:"timed_count"` + DurationP95MS *float64 `json:"duration_p95_ms"` +} + +type TurnMetricBucket struct { + Start time.Time `json:"start"` + Status string `json:"status"` + Count int64 `json:"count"` + QueueP95MS *float64 `json:"queue_p95_ms"` + ExecutionP95MS *float64 `json:"execution_p95_ms"` +} + +const requestMetricSelect = `SELECT + to_timestamp(floor(extract(epoch FROM bucket_start) / $3) * $3) AS bucket, + route_family, outcome, sum(request_count)::bigint, sum(latency_sum_ms)::bigint, + ARRAY[ + sum(latency_bucket_counts[1])::bigint, sum(latency_bucket_counts[2])::bigint, + sum(latency_bucket_counts[3])::bigint, sum(latency_bucket_counts[4])::bigint, + sum(latency_bucket_counts[5])::bigint, sum(latency_bucket_counts[6])::bigint, + sum(latency_bucket_counts[7])::bigint, sum(latency_bucket_counts[8])::bigint, + sum(latency_bucket_counts[9])::bigint, sum(latency_bucket_counts[10])::bigint, + sum(latency_bucket_counts[11])::bigint + ] AS latency_bucket_counts +FROM observability_request_minute_buckets +WHERE bucket_start >= $1 AND bucket_start < $2 +GROUP BY bucket, route_family, outcome ORDER BY bucket, route_family, outcome` + +const collectorMetricSelect = `SELECT + to_timestamp(floor(extract(epoch FROM bucket_start) / $3) * $3) AS bucket, + source, sum(attempted_count)::bigint, sum(observed_count)::bigint, + sum(unavailable_count)::bigint, sum(timeout_count)::bigint, + sum(dropped_count)::bigint, sum(export_failed_count)::bigint +FROM observability_collector_minute_buckets +WHERE bucket_start >= $1 AND bucket_start < $2 +GROUP BY bucket, source ORDER BY bucket, source` + +const turnMetricSelect = `SELECT + to_timestamp(floor(extract(epoch FROM completed_at) / $3) * $3) AS bucket, + status, count(*)::bigint, + (percentile_disc(0.95) WITHIN GROUP (ORDER BY greatest(extract(epoch FROM started_at - created_at) * 1000, 0)) + FILTER (WHERE started_at IS NOT NULL))::double precision, + (percentile_disc(0.95) WITHIN GROUP (ORDER BY greatest(extract(epoch FROM completed_at - started_at) * 1000, 0)) + FILTER (WHERE started_at IS NOT NULL))::double precision +FROM turns +WHERE completed_at >= $1 AND completed_at < $2 + AND status IN ('completed', 'failed', 'cancelled') +GROUP BY bucket, status ORDER BY bucket, status` + +const toolMetricSelect = `SELECT + to_timestamp(floor(extract(epoch FROM finished_at) / $3) * $3) AS bucket, + tool_category, outcome, count(*)::bigint, count(duration_ms)::bigint, + (percentile_disc(0.95) WITHIN GROUP (ORDER BY duration_ms) + FILTER (WHERE duration_ms IS NOT NULL))::double precision +FROM observability_tool_attempts +WHERE finished_at >= $1 AND finished_at < $2 +GROUP BY bucket, tool_category, outcome ORDER BY bucket, tool_category, outcome` + +// ReadOperatorMetrics returns only bounded labels and aggregate measurements. +func (s *Store) ReadOperatorMetrics(ctx context.Context, start, end time.Time, step time.Duration) (OperatorMetrics, error) { + result := OperatorMetrics{Requests: []RequestMetricBucket{}, Collector: []CollectorMetricBucket{}, Turns: []TurnMetricBucket{}, Tools: []ToolMetricBucket{}} + seconds := int64(step / time.Second) + rows, err := s.pool.Query(ctx, requestMetricSelect, start, end, seconds) + if err != nil { + return result, err + } + for rows.Next() { + var row RequestMetricBucket + if err := rows.Scan(&row.Start, &row.RouteFamily, &row.Outcome, &row.Count, &row.LatencySumMS, &row.LatencyCounts); err != nil { + rows.Close() + return result, err + } + result.Requests = append(result.Requests, row) + } + err = rows.Err() + rows.Close() + if err != nil { + return result, err + } + rows, err = s.pool.Query(ctx, collectorMetricSelect, start, end, seconds) + if err != nil { + return result, err + } + defer rows.Close() + for rows.Next() { + var row CollectorMetricBucket + if err := rows.Scan(&row.Start, &row.Source, &row.AttemptedCount, &row.ObservedCount, &row.UnavailableCount, + &row.TimeoutCount, &row.DroppedCount, &row.ExportFailedCount); err != nil { + return result, err + } + result.Collector = append(result.Collector, row) + } + if err := rows.Err(); err != nil { + return result, err + } + rows.Close() + rows, err = s.pool.Query(ctx, turnMetricSelect, start, end, seconds) + if err != nil { + return result, err + } + defer rows.Close() + for rows.Next() { + var row TurnMetricBucket + if err := rows.Scan(&row.Start, &row.Status, &row.Count, &row.QueueP95MS, &row.ExecutionP95MS); err != nil { + return result, err + } + result.Turns = append(result.Turns, row) + } + if err := rows.Err(); err != nil { + return result, err + } + rows.Close() + rows, err = s.pool.Query(ctx, toolMetricSelect, start, end, seconds) + if err != nil { + return result, err + } + defer rows.Close() + for rows.Next() { + var row ToolMetricBucket + if err := rows.Scan(&row.Start, &row.Category, &row.Outcome, &row.Count, &row.TimedCount, &row.DurationP95MS); err != nil { + return result, err + } + result.Tools = append(result.Tools, row) + } + return result, rows.Err() +} diff --git a/services/agents-api/internal/store/observability_requests.go b/services/agents-api/internal/store/observability_requests.go new file mode 100644 index 000000000..eebff6722 --- /dev/null +++ b/services/agents-api/internal/store/observability_requests.go @@ -0,0 +1,63 @@ +package store + +import ( + "context" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" +) + +const upsertRequestBucket = `INSERT INTO observability_request_minute_buckets AS current + (bucket_start, route_family, method, outcome, request_count, latency_sum_ms, latency_bucket_counts) +VALUES ($1, $2, $3, $4, $5, $6, $7) +ON CONFLICT (bucket_start, route_family, method, outcome) DO UPDATE SET + request_count = current.request_count + EXCLUDED.request_count, + latency_sum_ms = current.latency_sum_ms + EXCLUDED.latency_sum_ms, + latency_bucket_counts = ARRAY[ + current.latency_bucket_counts[1] + EXCLUDED.latency_bucket_counts[1], + current.latency_bucket_counts[2] + EXCLUDED.latency_bucket_counts[2], + current.latency_bucket_counts[3] + EXCLUDED.latency_bucket_counts[3], + current.latency_bucket_counts[4] + EXCLUDED.latency_bucket_counts[4], + current.latency_bucket_counts[5] + EXCLUDED.latency_bucket_counts[5], + current.latency_bucket_counts[6] + EXCLUDED.latency_bucket_counts[6], + current.latency_bucket_counts[7] + EXCLUDED.latency_bucket_counts[7], + current.latency_bucket_counts[8] + EXCLUDED.latency_bucket_counts[8], + current.latency_bucket_counts[9] + EXCLUDED.latency_bucket_counts[9], + current.latency_bucket_counts[10] + EXCLUDED.latency_bucket_counts[10], + current.latency_bucket_counts[11] + EXCLUDED.latency_bucket_counts[11] + ]` + +// WriteRequestBuckets persists asynchronous HTTP aggregates and collector coverage +// together. It is never called by the request handler's response path. +func (s *Store) WriteRequestBuckets(ctx context.Context, buckets []observability.RequestBucket, dropped, failed int64) error { + tx, err := s.pool.Begin(ctx) + if err != nil { + return err + } + defer tx.Rollback(ctx) + var written int64 + for _, bucket := range buckets { + if bucket.Count <= 0 { + continue + } + _, err = tx.Exec(ctx, upsertRequestBucket, bucket.Start, bucket.RouteFamily, bucket.Method, bucket.Outcome, + bucket.Count, bucket.LatencySumMS, bucket.LatencyCounts[:]) + if err != nil { + return err + } + written += bucket.Count + } + _, err = tx.Exec(ctx, `INSERT INTO observability_collector_minute_buckets AS current + (bucket_start, source, attempted_count, observed_count, dropped_count, export_failed_count) + VALUES ($1, 'request', $2, $3, $4, $5) + ON CONFLICT (bucket_start, source) DO UPDATE SET + attempted_count = current.attempted_count + EXCLUDED.attempted_count, + observed_count = current.observed_count + EXCLUDED.observed_count, + dropped_count = current.dropped_count + EXCLUDED.dropped_count, + export_failed_count = current.export_failed_count + EXCLUDED.export_failed_count`, + time.Now().UTC().Truncate(time.Minute), written+dropped, written, dropped, failed) + if err != nil { + return err + } + return tx.Commit(ctx) +} diff --git a/services/agents-api/internal/store/observability_requests_test.go b/services/agents-api/internal/store/observability_requests_test.go new file mode 100644 index 000000000..a886a65ba --- /dev/null +++ b/services/agents-api/internal/store/observability_requests_test.go @@ -0,0 +1,37 @@ +package store + +import ( + "testing" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" +) + +func TestPostgresOperatorRequestBuckets(t *testing.T) { + s, _ := testStore(t) + start := time.Now().UTC().Truncate(time.Minute).Add(-time.Minute) + first := observability.RequestBucket{Start: start, RouteFamily: "sessions", Method: "GET", Outcome: "success", Count: 2, LatencySumMS: 70} + first.LatencyCounts[1], first.LatencyCounts[2] = 1, 1 + second := observability.RequestBucket{Start: start, RouteFamily: "sessions", Method: "GET", Outcome: "success", Count: 1, LatencySumMS: 300} + second.LatencyCounts[5] = 1 + if err := s.WriteRequestBuckets(t.Context(), []observability.RequestBucket{first}, 0, 0); err != nil { + t.Fatal(err) + } + if err := s.WriteRequestBuckets(t.Context(), []observability.RequestBucket{second}, 1, 0); err != nil { + t.Fatal(err) + } + got, err := s.ReadOperatorMetrics(t.Context(), start, start.Add(time.Minute), time.Minute) + if err != nil { + t.Fatal(err) + } + if len(got.Requests) != 1 || got.Requests[0].Count != 3 || got.Requests[0].LatencySumMS != 370 || + got.Requests[0].LatencyCounts[1] != 1 || got.Requests[0].LatencyCounts[2] != 1 || got.Requests[0].LatencyCounts[5] != 1 { + t.Fatalf("request bucket lost concurrent-shape additions: %+v", got.Requests) + } + // Collector health is written in the current minute, separate from the + // completed request's event-time bucket. + health, err := s.ReadOperatorMetrics(t.Context(), start, start.Add(2*time.Minute), time.Minute) + if err != nil || len(health.Collector) == 0 { + t.Fatalf("collector health unavailable: %+v, %v", health.Collector, err) + } +} diff --git a/services/agents-api/internal/store/observability_retention.go b/services/agents-api/internal/store/observability_retention.go new file mode 100644 index 000000000..ff65fa4ea --- /dev/null +++ b/services/agents-api/internal/store/observability_retention.go @@ -0,0 +1,29 @@ +package store + +import ( + "context" + "time" +) + +// PruneOperatorMetrics deletes small batches so retention never holds a long lock. +func (s *Store) PruneOperatorMetrics(ctx context.Context, now time.Time) error { + cutoffs := []struct { + query string + before time.Time + }{ + {`DELETE FROM observability_request_minute_buckets WHERE ctid IN + (SELECT ctid FROM observability_request_minute_buckets WHERE bucket_start < $1 LIMIT 256 FOR UPDATE SKIP LOCKED)`, now.Add(-30 * 24 * time.Hour)}, + {`DELETE FROM observability_collector_minute_buckets WHERE ctid IN + (SELECT ctid FROM observability_collector_minute_buckets WHERE bucket_start < $1 LIMIT 256 FOR UPDATE SKIP LOCKED)`, now.Add(-30 * 24 * time.Hour)}, + {`DELETE FROM observability_model_attempts WHERE id IN + (SELECT id FROM observability_model_attempts WHERE started_at < $1 LIMIT 256 FOR UPDATE SKIP LOCKED)`, now.Add(-7 * 24 * time.Hour)}, + {`DELETE FROM observability_tool_attempts WHERE id IN + (SELECT id FROM observability_tool_attempts WHERE finished_at < $1 LIMIT 256 FOR UPDATE SKIP LOCKED)`, now.Add(-7 * 24 * time.Hour)}, + } + for _, cutoff := range cutoffs { + if _, err := s.pool.Exec(ctx, cutoff.query, cutoff.before); err != nil { + return err + } + } + return nil +} diff --git a/services/agents-api/internal/store/observability_tools.go b/services/agents-api/internal/store/observability_tools.go new file mode 100644 index 000000000..f6ef4f728 --- /dev/null +++ b/services/agents-api/internal/store/observability_tools.go @@ -0,0 +1,49 @@ +package store + +import ( + "context" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" +) + +const insertToolAttempt = `INSERT INTO observability_tool_attempts + (id, tenant_id, session_id, turn_id, started_at, finished_at, tool_category, outcome, duration_ms) +SELECT $1::uuid, s.tenant_id, s.id, t.id, $5::timestamptz, $6, $7, $8, $9::bigint +FROM sessions s JOIN turns t ON t.session_id = s.id +WHERE s.tenant_id = $2::uuid AND s.id = $3::uuid AND t.id = $4::uuid +ON CONFLICT (id) DO NOTHING` + +// WriteToolAttempts stores only sanitized terminal tool evidence. Duplicate +// native completion events have one deterministic attempt ID. +func (s *Store) WriteToolAttempts(ctx context.Context, attempts []observability.ToolAttempt, dropped, failed int64) error { + tx, err := s.pool.Begin(ctx) + if err != nil { + return err + } + defer tx.Rollback(ctx) + var inserted int64 + for _, attempt := range attempts { + result, err := tx.Exec(ctx, insertToolAttempt, attempt.ID, attempt.TenantID, attempt.SessionID, attempt.TurnID, + attempt.StartedAt, attempt.FinishedAt, attempt.Category, attempt.Outcome, attempt.DurationMS) + if err != nil { + return err + } + inserted += result.RowsAffected() + } + if len(attempts) > 0 || dropped > 0 || failed > 0 { + _, err = tx.Exec(ctx, `INSERT INTO observability_collector_minute_buckets AS current + (bucket_start, source, attempted_count, observed_count, dropped_count, export_failed_count) + VALUES ($1, 'tool', $2, $3, $4, $5) + ON CONFLICT (bucket_start, source) DO UPDATE SET + attempted_count = current.attempted_count + EXCLUDED.attempted_count, + observed_count = current.observed_count + EXCLUDED.observed_count, + dropped_count = current.dropped_count + EXCLUDED.dropped_count, + export_failed_count = current.export_failed_count + EXCLUDED.export_failed_count`, + time.Now().UTC().Truncate(time.Minute), int64(len(attempts))+dropped, inserted, dropped, failed) + if err != nil { + return err + } + } + return tx.Commit(ctx) +} diff --git a/services/agents-api/internal/store/observability_tools_test.go b/services/agents-api/internal/store/observability_tools_test.go new file mode 100644 index 000000000..a98c61a70 --- /dev/null +++ b/services/agents-api/internal/store/observability_tools_test.go @@ -0,0 +1,55 @@ +package store + +import ( + "testing" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/observability" + "github.com/google/uuid" +) + +func TestPostgresToolAttemptIsIdempotentAndTenantScoped(t *testing.T) { + s, pool := testStore(t) + tenant, session, turn, attemptID := uuid.NewString(), uuid.NewString(), uuid.NewString(), uuid.NewString() + _, err := pool.Exec(t.Context(), `INSERT INTO sessions (id, tenant_id, engine, idempotency_key, request_hash) + VALUES ($1, $2, 'codex', $3, 'observability-test')`, session, tenant, uuid.NewString()) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { + _, _ = pool.Exec(t.Context(), `DELETE FROM turns WHERE id=$1`, turn) + _, _ = pool.Exec(t.Context(), `DELETE FROM sessions WHERE id=$1`, session) + }) + _, err = pool.Exec(t.Context(), `INSERT INTO turns (id, session_id, status, completed_at) + VALUES ($1, $2, 'completed', clock_timestamp())`, turn, session) + if err != nil { + t.Fatal(err) + } + finished := time.Now().UTC() + duration := int64(120) + value := observability.ToolAttempt{ID: attemptID, TenantID: tenant, SessionID: session, TurnID: turn, + FinishedAt: finished, Category: "command", Outcome: "success", DurationMS: &duration} + if err := s.WriteToolAttempts(t.Context(), []observability.ToolAttempt{value}, 0, 0); err != nil { + t.Fatal(err) + } + if err := s.WriteToolAttempts(t.Context(), []observability.ToolAttempt{value}, 0, 0); err != nil { + t.Fatal(err) + } + var count int + var startedAt *time.Time + if err := pool.QueryRow(t.Context(), `SELECT count(*), max(started_at) FROM observability_tool_attempts WHERE id=$1`, attemptID).Scan(&count, &startedAt); err != nil || count != 1 || startedAt != nil { + t.Fatalf("tool attempt deduplication or unknown start failed: count=%d start=%v err=%v", count, startedAt, err) + } + value.ID = uuid.NewString() + value.TenantID = uuid.NewString() + if err := s.WriteToolAttempts(t.Context(), []observability.ToolAttempt{value}, 0, 0); err != nil { + t.Fatal(err) + } + if err := pool.QueryRow(t.Context(), `SELECT count(*) FROM observability_tool_attempts WHERE id=$1`, value.ID).Scan(&count); err != nil || count != 0 { + t.Fatalf("foreign tenant wrote an attempt: count=%d err=%v", count, err) + } + metrics, err := s.ReadOperatorMetrics(t.Context(), finished.Add(-time.Minute), finished.Add(time.Minute), time.Minute) + if err != nil || len(metrics.Tools) == 0 || len(metrics.Turns) == 0 { + t.Fatalf("terminal operator metrics unavailable: tools=%+v turns=%+v err=%v", metrics.Tools, metrics.Turns, err) + } +} diff --git a/services/agents-api/internal/store/runtime_history_acceptance_test.go b/services/agents-api/internal/store/runtime_history_acceptance_test.go index dcedf2c99..fc26a2b16 100644 --- a/services/agents-api/internal/store/runtime_history_acceptance_test.go +++ b/services/agents-api/internal/store/runtime_history_acceptance_test.go @@ -148,10 +148,10 @@ func TestPostgresRuntimeHistoryDenseReadAndBoundedRetention(t *testing.T) { } // Seed the equivalent of a full day of five-second periodic samples in one // fixture statement; production inserts still go through the exporter. - _, err := pool.Exec(t.Context(), `INSERT INTO runtime_history_samples + _, err := pool.Exec(t.Context(), `INSERT INTO observability_runtime_samples (tenant_id,session_id,environment_id,resolved_at_ns,allocation_id,provider_type,status,observed_at_ns,started_at_ns,cpu_usage_seconds,cpu_capacity_cores,memory_usage_bytes,input_tokens,output_tokens) SELECT tenant_id,session_id,environment_id,resolved_at_ns+n*5000000000,allocation_id,provider_type,status,observed_at_ns+n*5000000000,started_at_ns,n::double precision,cpu_capacity_cores,memory_usage_bytes,input_tokens+n,output_tokens+n - FROM runtime_history_samples CROSS JOIN generate_series(1,17279) n WHERE tenant_id=$1 AND session_id=$2`, scope.TenantID, scope.SessionID) + FROM observability_runtime_samples CROSS JOIN generate_series(1,17279) n WHERE tenant_id=$1 AND session_id=$2`, scope.TenantID, scope.SessionID) if err != nil { t.Fatal(err) } @@ -173,10 +173,10 @@ func TestPostgresRuntimeHistoryDenseReadAndBoundedRetention(t *testing.T) { if err := s.InsertRuntimeHistorySample(t.Context(), old); err != nil { t.Fatal(err) } - _, err = pool.Exec(t.Context(), `INSERT INTO runtime_history_samples + _, err = pool.Exec(t.Context(), `INSERT INTO observability_runtime_samples (tenant_id,session_id,environment_id,resolved_at_ns,allocation_id,provider_type,status,observed_at_ns,started_at_ns) SELECT tenant_id,session_id,environment_id,resolved_at_ns-n,allocation_id,provider_type,'unavailable',NULL,NULL - FROM runtime_history_samples CROSS JOIN generate_series(1,4100) n WHERE tenant_id=$1 AND session_id=$2 AND resolved_at_ns=$3`, scope.TenantID, scope.SessionID, old.ResolvedAt.UnixNano()) + FROM observability_runtime_samples CROSS JOIN generate_series(1,4100) n WHERE tenant_id=$1 AND session_id=$2 AND resolved_at_ns=$3`, scope.TenantID, scope.SessionID, old.ResolvedAt.UnixNano()) if err != nil { t.Fatal(err) } @@ -188,13 +188,13 @@ func TestPostgresRuntimeHistoryDenseReadAndBoundedRetention(t *testing.T) { t.Fatal(err) } var remaining int - if err := pool.QueryRow(t.Context(), `SELECT count(*) FROM runtime_history_samples WHERE tenant_id=$1 AND resolved_at_ns<$2`, scope.TenantID, end.Add(-7*24*time.Hour).UnixNano()).Scan(&remaining); err != nil || remaining != 5 { + if err := pool.QueryRow(t.Context(), `SELECT count(*) FROM observability_runtime_samples WHERE tenant_id=$1 AND resolved_at_ns<$2`, scope.TenantID, end.Add(-7*24*time.Hour).UnixNano()).Scan(&remaining); err != nil || remaining != 5 { t.Fatal("cleanup exceeded bounded batch", remaining, err) } if _, err := reader.Prune(t.Context()); err != nil { t.Fatal(err) } - if err := pool.QueryRow(t.Context(), `SELECT count(*) FROM runtime_history_samples WHERE tenant_id=$1 AND resolved_at_ns<$2`, scope.TenantID, end.Add(-7*24*time.Hour).UnixNano()).Scan(&remaining); err != nil || remaining != 0 { + if err := pool.QueryRow(t.Context(), `SELECT count(*) FROM observability_runtime_samples WHERE tenant_id=$1 AND resolved_at_ns<$2`, scope.TenantID, end.Add(-7*24*time.Hour).UnixNano()).Scan(&remaining); err != nil || remaining != 0 { t.Fatal("retention cleanup incomplete", remaining, err) } cancelled, cancel := context.WithCancel(t.Context()) diff --git a/services/agents-api/migrations/000069_observability.sql b/services/agents-api/migrations/000069_observability.sql new file mode 100644 index 000000000..263db79d3 --- /dev/null +++ b/services/agents-api/migrations/000069_observability.sql @@ -0,0 +1,78 @@ +-- +goose Up +-- Runtime samples remain a bounded telemetry projection, not execution or Usage authority. +ALTER TABLE runtime_history_samples RENAME TO observability_runtime_samples; +ALTER INDEX runtime_history_samples_pkey RENAME TO observability_runtime_samples_pkey; +ALTER INDEX runtime_history_samples_retention_idx RENAME TO observability_runtime_samples_retention_idx; +CREATE INDEX turns_completed_time_idx ON turns (completed_at) WHERE completed_at IS NOT NULL; + +-- Histogram bins: <=10, 25, 50, 100, 250, 500, 1000, 2500, 5000, +-- 10000 milliseconds, and >10000 milliseconds. Counts are disjoint. +CREATE TABLE observability_request_minute_buckets ( + bucket_start timestamptz NOT NULL, + route_family text NOT NULL CHECK (length(route_family) BETWEEN 1 AND 64), + method text NOT NULL CHECK (method IN ('GET', 'POST', 'PUT', 'PATCH', 'DELETE', 'HEAD', 'OTHER')), + outcome text NOT NULL CHECK (outcome IN ('success', 'client_error', 'server_error')), + request_count bigint NOT NULL DEFAULT 0 CHECK (request_count >= 0), + latency_sum_ms bigint NOT NULL DEFAULT 0 CHECK (latency_sum_ms >= 0), + latency_bucket_counts bigint[] NOT NULL DEFAULT array_fill(0::bigint, ARRAY[11]) + CHECK (array_length(latency_bucket_counts, 1) = 11), + PRIMARY KEY (bucket_start, route_family, method, outcome) +); + +CREATE TABLE observability_model_attempts ( + id uuid PRIMARY KEY, + tenant_id uuid NOT NULL, + session_id uuid NOT NULL REFERENCES sessions(id) ON DELETE CASCADE, + turn_id uuid NOT NULL REFERENCES turns(id) ON DELETE CASCADE, + started_at timestamptz NOT NULL, + finished_at timestamptz NOT NULL CHECK (finished_at >= started_at), + model_family text NOT NULL CHECK (length(model_family) BETWEEN 1 AND 64), + provider_type text NOT NULL CHECK (length(provider_type) BETWEEN 1 AND 64), + outcome text NOT NULL CHECK (outcome IN ('success', 'error', 'cancelled', 'unknown')), + duration_ms bigint NOT NULL CHECK (duration_ms >= 0), + input_tokens bigint CHECK (input_tokens >= 0), + output_tokens bigint CHECK (output_tokens >= 0), + usage_status text NOT NULL CHECK (usage_status IN ('observed', 'unavailable', 'unsupported')), + CHECK ((input_tokens IS NULL AND output_tokens IS NULL) OR usage_status = 'observed') +); +CREATE INDEX observability_model_attempts_tenant_time_idx ON observability_model_attempts (tenant_id, started_at DESC); +CREATE INDEX observability_model_attempts_session_time_idx ON observability_model_attempts (session_id, started_at DESC); +CREATE INDEX observability_model_attempts_retention_idx ON observability_model_attempts (started_at); + +CREATE TABLE observability_tool_attempts ( + id uuid PRIMARY KEY, + tenant_id uuid NOT NULL, + session_id uuid NOT NULL REFERENCES sessions(id) ON DELETE CASCADE, + turn_id uuid NOT NULL REFERENCES turns(id) ON DELETE CASCADE, + started_at timestamptz, + finished_at timestamptz NOT NULL, + tool_category text NOT NULL CHECK (length(tool_category) BETWEEN 1 AND 64), + outcome text NOT NULL CHECK (outcome IN ('success', 'error', 'cancelled', 'unknown')), + duration_ms bigint CHECK (duration_ms >= 0), + CHECK (started_at IS NULL OR finished_at >= started_at) +); +CREATE INDEX observability_tool_attempts_tenant_time_idx ON observability_tool_attempts (tenant_id, finished_at DESC); +CREATE INDEX observability_tool_attempts_session_time_idx ON observability_tool_attempts (session_id, finished_at DESC); +CREATE INDEX observability_tool_attempts_retention_idx ON observability_tool_attempts (finished_at); + +CREATE TABLE observability_collector_minute_buckets ( + bucket_start timestamptz NOT NULL, + source text NOT NULL CHECK (source IN ('runtime', 'request', 'model', 'tool', 'export')), + attempted_count bigint NOT NULL DEFAULT 0 CHECK (attempted_count >= 0), + observed_count bigint NOT NULL DEFAULT 0 CHECK (observed_count >= 0), + unavailable_count bigint NOT NULL DEFAULT 0 CHECK (unavailable_count >= 0), + timeout_count bigint NOT NULL DEFAULT 0 CHECK (timeout_count >= 0), + dropped_count bigint NOT NULL DEFAULT 0 CHECK (dropped_count >= 0), + export_failed_count bigint NOT NULL DEFAULT 0 CHECK (export_failed_count >= 0), + PRIMARY KEY (bucket_start, source) +); + +-- +goose Down +DROP TABLE observability_collector_minute_buckets; +DROP TABLE observability_tool_attempts; +DROP TABLE observability_model_attempts; +DROP TABLE observability_request_minute_buckets; +DROP INDEX turns_completed_time_idx; +ALTER INDEX observability_runtime_samples_retention_idx RENAME TO runtime_history_samples_retention_idx; +ALTER INDEX observability_runtime_samples_pkey RENAME TO runtime_history_samples_pkey; +ALTER TABLE observability_runtime_samples RENAME TO runtime_history_samples; diff --git a/services/agents-api/tests/runtime_history_live.py b/services/agents-api/tests/runtime_history_live.py index ba06f92fc..3e0c44017 100644 --- a/services/agents-api/tests/runtime_history_live.py +++ b/services/agents-api/tests/runtime_history_live.py @@ -33,7 +33,7 @@ def request(self, *args, **kwargs): encoded = json.dumps(value) base.require(not any(secret in encoded for secret in self.secrets), "public_response_exposed_secret") base.require(not any(marker in encoded for marker in ( - "postgres://", "postgresql://", "runtime_history_samples", "token_sha256", + "postgres://", "postgresql://", "observability_runtime_samples", "token_sha256", "unix:///", "host.microsandbox.internal", "/home/parsar-acceptance/")), "public_response_exposed_backend_configuration") return value diff --git a/services/core-console/admin_routes.go b/services/core-console/admin_routes.go index 7c2d6d410..fce80e66b 100644 --- a/services/core-console/admin_routes.go +++ b/services/core-console/admin_routes.go @@ -26,7 +26,7 @@ func adminAPIRequest(r *http.Request) bool { return read || r.Method == http.MethodPost case "copies": return r.Method == http.MethodPost - case "summary", "runtime-observations", "startup-configuration", "audit-log": + case "summary", "runtime-observations", "startup-configuration", "audit-log", "core-metrics", "observability": return read } } diff --git a/services/core-console/sandbox_admin_test.go b/services/core-console/sandbox_admin_test.go index 2f6c74a2d..60a8836cd 100644 --- a/services/core-console/sandbox_admin_test.go +++ b/services/core-console/sandbox_admin_test.go @@ -27,7 +27,7 @@ func TestSandboxAdminUsesAuthenticatedConsoleAndServerCredential(t *testing.T) { } w.WriteHeader(204) })) - for _, tc := range []struct{ method, path string }{{"GET", "/core/v1/sandbox/deployment"}, {"GET", "/core/v1/sandbox/nodes?limit=5"}, {"PATCH", "/core/v1/sandbox/nodes/node"}, {"DELETE", "/core/v1/sandbox/nodes/node"}, {"GET", "/core/v1/sandbox/nodes/node/allocations"}, {"POST", "/core/v1/sandbox/enrollment-tokens"}} { + for _, tc := range []struct{ method, path string }{{"GET", "/core/v1/sandbox/deployment"}, {"GET", "/core/v1/sandbox/nodes?limit=5"}, {"PATCH", "/core/v1/sandbox/nodes/node"}, {"DELETE", "/core/v1/sandbox/nodes/node"}, {"GET", "/core/v1/sandbox/nodes/node/allocations"}, {"POST", "/core/v1/sandbox/enrollment-tokens"}, {"GET", "/core/v1/admin/observability?range=1h"}, {"GET", "/core/v1/admin/core-metrics?range=1h"}} { req := consoleRequest(t, server, tc.method, tc.path) req.Header.Set("X-Core-Console-Actor", "spoofed") response, _ := responseBody(t, server, req) @@ -41,10 +41,30 @@ func TestSandboxAdminUsesAuthenticatedConsoleAndServerCredential(t *testing.T) { t.Fatal("bearer bypassed console login") } } - if calls.Load() != 6 { + if calls.Load() != 8 { t.Fatal("unexpected sandbox proxy calls") } } + +func TestObservabilityAdminAllowsOnlyFixedRead(t *testing.T) { + var calls atomic.Int32 + server, _ := testConsole(t, http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + calls.Add(1) + w.WriteHeader(http.StatusNoContent) + })) + for _, tc := range []struct{ method, path string }{ + {"POST", "/core/v1/admin/observability"}, + {"GET", "/core/v1/admin/observability/extra"}, + } { + response, _ := responseBody(t, server, consoleRequest(t, server, tc.method, tc.path)) + if response.StatusCode != http.StatusNotFound { + t.Errorf("%s %s = %d", tc.method, tc.path, response.StatusCode) + } + } + if calls.Load() != 0 { + t.Fatal("unsupported observability path reached Core") + } +} func TestSandboxAdminKeepsOriginAndPathBoundary(t *testing.T) { server, _ := testConsole(t, http.HandlerFunc(func(http.ResponseWriter, *http.Request) { t.Error("unsafe request reached Core") })) for _, tc := range []struct {