From 9289f4ab7252e19ffc8bbf30950da98c0f2473f0 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Sat, 12 Sep 2026 13:10:49 -0300 Subject: [PATCH 01/30] docs(spec): add design docs for regression commit attribution --- .../checklists/requirements.md | 35 +++ .../contracts/report-and-cockpit.md | 83 +++++++ .../contracts/results-json.md | 70 ++++++ .../data-model.md | 95 ++++++++ .../012-regression-commit-attribution/plan.md | 178 +++++++++++++++ .../quickstart.md | 92 ++++++++ .../research.md | 153 +++++++++++++ .../012-regression-commit-attribution/spec.md | 113 ++++++++++ .../tasks.md | 211 ++++++++++++++++++ 9 files changed, 1030 insertions(+) create mode 100644 specs/012-regression-commit-attribution/checklists/requirements.md create mode 100644 specs/012-regression-commit-attribution/contracts/report-and-cockpit.md create mode 100644 specs/012-regression-commit-attribution/contracts/results-json.md create mode 100644 specs/012-regression-commit-attribution/data-model.md create mode 100644 specs/012-regression-commit-attribution/plan.md create mode 100644 specs/012-regression-commit-attribution/quickstart.md create mode 100644 specs/012-regression-commit-attribution/research.md create mode 100644 specs/012-regression-commit-attribution/spec.md create mode 100644 specs/012-regression-commit-attribution/tasks.md diff --git a/specs/012-regression-commit-attribution/checklists/requirements.md b/specs/012-regression-commit-attribution/checklists/requirements.md new file mode 100644 index 00000000..9f16dc8e --- /dev/null +++ b/specs/012-regression-commit-attribution/checklists/requirements.md @@ -0,0 +1,35 @@ +# Specification Quality Checklist: Regression Commit Attribution + +**Purpose**: Validate specification completeness and quality before proceeding to planning +**Created**: 2026-09-11 +**Feature**: [spec.md](../spec.md) + +## Content Quality + +- [x] No implementation details (languages, frameworks, APIs) +- [x] Focused on user value and business needs +- [x] Written for non-technical stakeholders +- [x] All mandatory sections completed + +## Requirement Completeness + +- [x] No [NEEDS CLARIFICATION] markers remain +- [x] Requirements are testable and unambiguous +- [x] Success criteria are measurable +- [x] Success criteria are technology-agnostic (no implementation details) +- [x] All acceptance scenarios are defined +- [x] Edge cases are identified +- [x] Scope is clearly bounded +- [x] Dependencies and assumptions identified + +## Feature Readiness + +- [x] All functional requirements have clear acceptance criteria +- [x] User scenarios cover primary flows +- [x] Feature meets measurable outcomes defined in Success Criteria +- [x] No implementation details leak into specification + +## Notes + +- Reasonable defaults (reuse of existing methodology-fingerprint comparability, existing regression-detection thresholds, CI-provided commit SHA availability) are documented in the Assumptions section rather than raised as clarification questions, per product direction already given (prioritize the hosted-agent/CI scenario) in the source conversation. +- All items pass; no follow-up needed before `/speckit-plan`. diff --git a/specs/012-regression-commit-attribution/contracts/report-and-cockpit.md b/specs/012-regression-commit-attribution/contracts/report-and-cockpit.md new file mode 100644 index 00000000..ffb82e85 --- /dev/null +++ b/specs/012-regression-commit-attribution/contracts/report-and-cockpit.md @@ -0,0 +1,83 @@ +# Contract: `report.md` section and Cockpit history payload + +## `report.md`: new "Regression Insight" section + +Rendered by `pipeline/reporter.py` immediately after the existing "Comparison +vs Baseline" section (`_render_comparison`), and only when +`result.comparison.insight` is present: + +```markdown +## Comparison vs Baseline +...existing table... + +## Regression Insight + +Run v3 → v4: **accuracy** dropped from 0.91 to 0.79. + +**Likely cause:** the system prompt changed and the model changed from +`gpt-4o` to `gpt-4o-mini`. + +**Suggested action:** review the prompt change and the model swap; consider +reverting one at a time to isolate the cause. +``` + +This section is purely additive to the report: it never replaces the +existing Metrics/Thresholds/Comparison/Rows sections, and its absence (no +regression, or missing commit metadata on either side) leaves `report.md` +byte-for-byte identical to today's output. Because this is the same +`report.md` already uploaded as the `agentops-pr-results` CI artifact and +posted as the PR comment by the generated `agentops-pr.yml` workflow, no +workflow template changes are required to deliver it (FR-010). + +Doctor's rolling-baseline `Finding` (`agent/checks/regression.py`) gets the +same explanation text via `Finding.evidence["insight"]`, and Doctor's own +Markdown rendering already surfaces `Finding.summary`/`recommendation` — this +plan extends that finding's `recommendation` text with the same +deterministic explanation when an insight is available, in place of today's +generic "inspect prompt/model/dataset changes" instruction. + +## Cockpit: version history + +No new HTTP route. The existing partial-load endpoint +(`GET /?_partial=1`, `cockpit.py:5405`) response gains a new section +alongside the existing eval "cards" built by `_build_eval_section` +(`cockpit.py:180`): + +```jsonc +{ + // ...existing cockpit payload sections unchanged... + "eval_history": { + "has_runs": true, + "entries": [ + { + "run_id": "20260901-101500", + "timestamp": "2026-09-01T10:15:00Z", + "commit": { "short_sha": "a1b2c3d", "subject": "..." }, + "metrics": { "accuracy": 0.91 }, + "methodology_fingerprint": "9f2a...", + "changed_inputs": [], + "regressed": false + }, + { + "run_id": "20260910-140300", + "timestamp": "2026-09-10T14:03:00Z", + "commit": { "short_sha": "b7e91aa", "subject": "..." }, + "metrics": { "accuracy": 0.79 }, + "methodology_fingerprint": "9f2a...", + "changed_inputs": [ + { "field": "model", "description": "model changed from gpt-4o to gpt-4o-mini" } + ], + "regressed": true + } + ] + } +} +``` + +Entries are grouped/ordered by `methodology_fingerprint` (same grouping key +already used by `results_history.py` and Doctor's regression check), oldest +first. A run with no resolvable `commit` still appears, with `commit: null` +and an empty `changed_inputs` list rather than being omitted (per User Story +2's acceptance scenario 3). This view is read-only and introduces no writes +to any monitored resource, consistent with the constitution's Cockpit +read-only requirement. diff --git a/specs/012-regression-commit-attribution/contracts/results-json.md b/specs/012-regression-commit-attribution/contracts/results-json.md new file mode 100644 index 00000000..eb756624 --- /dev/null +++ b/specs/012-regression-commit-attribution/contracts/results-json.md @@ -0,0 +1,70 @@ +# Contract: `results.json` additions + +`results.json` is a documented public contract (constitution, Principle I). +This feature only adds optional, additive fields — no existing field +changes meaning or type, and no consumer that ignores unknown-to-it new +top-level keys breaks. + +## New top-level field: `commit` + +```jsonc +{ + // ...existing RunResult fields unchanged... + "commit": { + "sha": "a1b2c3d4e5f6...", + "short_sha": "a1b2c3d", + "subject": "Swap eval agent to gpt-4o-mini", + "author": "Jane Doe", + "authored_at": "2026-09-10T14:03:00Z", + "source": "ci" + } + // or "commit": null when it could not be determined +} +``` + +Absent/`null` on any run produced before this feature shipped, or when commit +metadata could not be resolved (not a git repo, git unavailable). Consumers +MUST treat a missing/`null` `commit` the same as before this feature existed. + +## New field on `comparison`: `insight` + +Only present when `comparison` (the existing `--baseline` block) is present +**and** a regression was detected **and** both the current and baseline runs +have non-null `commit`: + +```jsonc +{ + "comparison": { + // ...existing ComparisonInfo fields unchanged... + "insight": { + "from_run_id": "20260901-101500", + "to_run_id": "20260910-140300", + "from_commit": { "sha": "...", "short_sha": "...", "subject": "...", "author": "...", "authored_at": "...", "source": "ci" }, + "to_commit": { "sha": "...", "short_sha": "...", "subject": "...", "author": "...", "authored_at": "...", "source": "ci" }, + "metric": "accuracy", + "from_value": 0.91, + "to_value": 0.79, + "changed_inputs": [ + { "field": "system_prompt", "description": "the system prompt changed", "from_value": "greeter:v3", "to_value": "greeter:v4" }, + { "field": "model", "description": "model changed from gpt-4o to gpt-4o-mini", "from_value": "gpt-4o", "to_value": "gpt-4o-mini" } + ], + "explanation": "Run v3 → v4: accuracy dropped from 0.91 to 0.79. Likely cause: the system prompt changed and the model changed from gpt-4o to gpt-4o-mini.", + "suggested_action": "Review the prompt change and the model swap; consider reverting one at a time to isolate the cause.", + "used_git_diff": false + } + } +} +``` + +Absent (`comparison.insight` key not present) whenever no regression was +detected, either compared run lacks commit metadata, or `comparison` itself +is absent (no `--baseline` was used). This preserves every existing +`comparison`-shaped consumer. + +## No changes to exit codes or CLI flags + +This feature introduces no new `agentops eval run` flags and does not change +the exit-code contract (`0`/`1`/`2` meanings are unchanged, per FR-011). The +committed-baseline auto-detection already used by generated PR workflows +(`.agentops/baseline/results.json` → `--baseline`) is unchanged; this feature +only adds richer content to what that path already produces. diff --git a/specs/012-regression-commit-attribution/data-model.md b/specs/012-regression-commit-attribution/data-model.md new file mode 100644 index 00000000..3c02195d --- /dev/null +++ b/specs/012-regression-commit-attribution/data-model.md @@ -0,0 +1,95 @@ +# Phase 1 Data Model: Regression Commit Attribution + +All new shapes are additive Pydantic v2 models in `src/agentops/core/results.py` +(per the constitution's "Preserve Public Contracts" and "pure `core/`" rules — +no I/O in these models). Existing fields are unchanged; nothing is removed. + +## CommitInfo + +Represents the git commit a single evaluation run was produced from. + +| Field | Type | Notes | +|---|---|---| +| `sha` | `str` | Full commit SHA. Required when the object exists at all. | +| `short_sha` | `str` | First 7-12 chars, for display. | +| `subject` | `str` | First line of the commit message. | +| `author` | `str` | Commit author name (not committer). | +| `authored_at` | `str` | ISO 8601 commit timestamp. | +| `source` | `Literal["ci", "local"]` | How the SHA was resolved (CI env var vs. `git rev-parse HEAD`). | + +**Validation rule**: `RunResult.commit` is either a fully-populated +`CommitInfo` or `None` — never a partial object. If any required piece +(subject/author/timestamp via `git show`) cannot be resolved after a SHA was +found, `commit` is left `None` rather than persisting a half-filled record. + +## ChangedInput + +One concrete difference detected between two runs' recorded fields. + +| Field | Type | Notes | +|---|---|---| +| `field` | `str` | One of `"system_prompt"`, `"model"`, `"dataset"`, `"evaluators"`, `"thresholds"`, or another tracked config key. | +| `description` | `str` | Human-readable one-liner, e.g. `"model changed from gpt-4o to gpt-4o-mini"`. | +| `from_value` | `Optional[str]` | Prior value (truncated/hashed for large values such as prompt version identifiers). | +| `to_value` | `Optional[str]` | New value. | + +## RegressionInsight + +The causal explanation for a detected regression between two comparable runs. + +| Field | Type | Notes | +|---|---|---| +| `from_run_id` | `str` | Identifier of the prior comparable run. | +| `to_run_id` | `str` | Identifier of the regressed run. | +| `from_commit` | `Optional[CommitInfo]` | Prior run's commit, if known. | +| `to_commit` | `Optional[CommitInfo]` | Regressed run's commit, if known. | +| `metric` | `str` | The regressed metric's name. | +| `from_value` | `float` | Metric value on the prior run. | +| `to_value` | `float` | Metric value on the regressed run. | +| `changed_inputs` | `List[ChangedInput]` | All detected changes, not just the first found (per spec edge case). | +| `explanation` | `str` | The rendered one-to-two-sentence plain-language summary (FR-007). | +| `suggested_action` | `Optional[str]` | Brief, rule-based corrective suggestion (FR-008). | +| `used_git_diff` | `bool` | `True` when both commits were reachable in local git history for a fuller diff; `False` when this fell back to comparing only the fields already recorded in each run's stored result (research.md #4). | + +**Validation rule**: A `RegressionInsight` is only ever constructed when both +`from_commit` and `to_commit` are non-`None` (FR-006, FR-009). If either run +lacks commit metadata, no `RegressionInsight` is produced at all — the plain +metric comparison (already existing `ComparisonInfo`/`Finding`) is left +untouched. + +## Extensions to existing models + +- **`RunResult`** (`src/agentops/core/results.py:100`): add + `commit: Optional[CommitInfo] = None`. Fully additive; `model_config = + ConfigDict(extra="forbid")` is preserved by declaring the field rather than + stuffing it into the free-form `config: Dict[str, Any]`. +- **`ComparisonInfo`** (`results.py:90`): add + `insight: Optional[RegressionInsight] = None`, populated by + `pipeline/comparison.py` (or the caller in `orchestrator.py`) when a + regressed metric and both commits are available. +- **Doctor `Finding.evidence`** (`agent/findings.py`, already + `Dict[str, Any]`): gains an optional `"insight"` key holding + `RegressionInsight.model_dump()` when `agent/checks/regression.py` produces + one — no schema change needed since `evidence` is already free-form. + +## Cockpit projection (not a persisted model — computed per request) + +**VersionHistoryEntry** (dict shape returned by `_project_run()` / +consumed by the cockpit UI): + +| Key | Type | Notes | +|---|---|---| +| `run_id` | `str` | Existing field. | +| `timestamp` | `Optional[str]` | Existing field. | +| `commit` | `Optional[dict]` | New: `CommitInfo.model_dump()` when known. | +| `metrics` | `Dict[str, float]` | Existing field. | +| `methodology_fingerprint` | `Optional[str]` | New: reused from `results_history._methodology_fingerprint()` so entries can be grouped/ordered per lineage. | +| `changed_inputs` | `List[dict]` | New: `ChangedInput` list vs. the previous entry sharing the same fingerprint; empty for the first run of a fingerprint. | +| `regressed` | `bool` | New: whether any metric regressed vs. the previous entry (drives a visual marker; the list itself is not regression-gated per FR-012/User Story 2). | + +## State / lifecycle notes + +These are all point-in-time computed values attached to an immutable run +result at write time (`commit`) or derived at read time (`insight`, +`changed_inputs`) — there are no state transitions to model. A `RunResult`, +once persisted, is never mutated in place. diff --git a/specs/012-regression-commit-attribution/plan.md b/specs/012-regression-commit-attribution/plan.md new file mode 100644 index 00000000..a8021017 --- /dev/null +++ b/specs/012-regression-commit-attribution/plan.md @@ -0,0 +1,178 @@ +# Implementation Plan: Regression Commit Attribution + +**Branch**: `012-regression-commit-attribution` | **Date**: 2026-09-11 | **Spec**: [spec.md](./spec.md) + +**Input**: Feature specification from `/specs/012-regression-commit-attribution/spec.md` + +**Note**: This template is filled in by the `/speckit-plan` command; its definition describes the execution workflow. + +## Summary + +When an evaluation run's metrics regress relative to the previous comparable +run, nobody is told why automatically today — someone has to manually diff +prompts, model choice, and config between two commits. This feature (1) +records commit metadata (SHA, subject, author, timestamp) on every evaluation +run's stored result, prioritizing the Foundry hosted-agent / cloud-CI scenario +where CI already reliably exposes the commit SHA, with best-effort local-git +capture as a secondary path; and (2) when two comparable runs both have +commit metadata and a metric regressed between them, deterministically diffs +their recorded prompt/model/config fields and produces a plain-language, +one-to-two-sentence explanation plus a brief suggested fix. The explanation +is surfaced in the existing `report.md` (already attached to the PR pipeline +as a CI artifact and PR comment — no new delivery mechanism) and in a new +Cockpit version-history view listing every evaluated run and what changed +relative to the previous one, independent of regression. + +Technical approach: a single new commit-metadata resolver +(`pipeline/commit_info.py`) attached at the one existing choke point every +execution mode already shares (`orchestrator._persist()`), plus a single +deterministic diff/explanation helper (`pipeline/regression_insight.py`) +consumed by both existing regression signals (`pipeline/comparison.py`'s +`--baseline` comparison and Doctor's rolling-baseline +`agent/checks/regression.py`) and by Cockpit's existing run-history scan +(`agent/cockpit.py`). All additions are additive to `results.json` and +introduce no new CLI flags, Azure calls, or exit-code semantics. + +## Technical Context + +**Language/Version**: Python 3.11+ (per constitution's supported runtime) + +**Primary Dependencies**: None new. Reuses stdlib `subprocess` (for `git +show`/`git rev-parse`, mirroring the existing pattern in +`pipeline/prompt_deploy.py`), and existing Pydantic v2 models in `core/`. + +**Storage**: Flat JSON files (existing) — `.agentops/results//results.json` +(+ `latest/` mirror) and the optional committed +`.agentops/baseline/results.json`. This feature adds fields to that existing +schema; it introduces no database and no new artifact file. + +**Testing**: pytest, per constitution Principle V — new unit tests under +`tests/unit/` for the commit resolver, the diff/explanation helper, reporter +rendering, and Cockpit projection; one integration test under +`tests/integration/` for the end-to-end two-run regression scenario. + +**Target Platform**: Same as today — local developer machines (macOS/Linux/ +Windows) and CI runners (GitHub Actions, Azure DevOps) executing the +`agentops` CLI. No server component beyond the existing local Cockpit. + +**Project Type**: Single project (existing `src/agentops/` CLI + services +layout); no frontend/backend split. + +**Performance Goals**: Negligible added overhead per run — a small, fixed +number of local `git` subprocess calls (sub-100ms typically) and in-memory +field comparisons against at most one prior run; must not measurably slow +`agentops eval run` or CI job duration. + +**Constraints**: No new Azure SDK calls (git operations are local +subprocess only, consistent with Principle III's lazy/isolated Azure +integration rule not even applying here); must not change the exit-code or +threshold-gating contract (FR-011); must degrade to today's behavior with no +error when git is unavailable, the workspace isn't a repo, history is +shallow/pruned, or either compared run lacks commit metadata (FR-003, FR-009). + +**Scale/Scope**: Per-workspace local run history (tens to low hundreds of +`.agentops/results/*` entries) and a small, fixed number of git commands per +run; no pagination or indexing concerns at this scale. + +## Constitution Check + +*GATE: Must pass before Phase 0 research. Re-check after Phase 1 design.* + +- **I. Preserve Public Contracts** — PASS. All `results.json` changes are + additive optional fields (`commit`, `comparison.insight`); no existing + field's type or meaning changes; no new CLI flags; exit-code contract + (`0`/`1`/`2`) is explicitly unchanged (FR-011). See + `contracts/results-json.md`. +- **II. Enforce Architectural Boundaries** — PASS. New git-subprocess and + diff logic lives in `pipeline/` (`commit_info.py`, `regression_insight.py`); + Cockpit-specific rendering stays in `agent/cockpit.py`; `core/results.py` + gains only pure, I/O-free data models (`CommitInfo`, `ChangedInput`, + `RegressionInsight`). `cli/app.py` is untouched — no new parsing/output + responsibilities added there. +- **III. Isolate Azure Runtime Integration** — PASS (not applicable to the + new code paths). No Azure SDK calls are introduced; commit capture and + diffing are local-git-only and function identically with or without Azure + credentials. +- **IV. Keep Release Evidence Trustworthy** — PASS. Cockpit's new history + view is read-only, reusing the existing local-file scan; Doctor's + regression `Finding` gains a richer `recommendation` string but no new + exit-code contract (`evidence` is already free-form). No monitored cloud + resource is mutated or deleted. +- **V. Verify Every Behavior Change** — PASS (planned). Unit coverage for + the resolver, the diff helper (including both no-metadata and + no-git-history fallback branches), reporter rendering, and Cockpit + projection; one integration test for the end-to-end regression scenario. + See `quickstart.md`'s "Automated coverage" section; concrete test tasks are + generated in `tasks.md` by `/speckit-tasks`. + +No violations requiring an entry in Complexity Tracking. + +**Post-design re-check (after Phase 1)**: Unchanged — the data model +(`data-model.md`) and contracts (`contracts/`) confirmed above stay purely +additive and introduce no new architectural layer, Azure dependency, or +public-contract break. Gate still PASSES. + +## Project Structure + +### Documentation (this feature) + +```text +specs/012-regression-commit-attribution/ +├── plan.md # This file (/speckit-plan command output) +├── research.md # Phase 0 output (/speckit-plan command) +├── data-model.md # Phase 1 output (/speckit-plan command) +├── quickstart.md # Phase 1 output (/speckit-plan command) +├── contracts/ # Phase 1 output (/speckit-plan command) +│ ├── results-json.md +│ └── report-and-cockpit.md +├── checklists/ +│ └── requirements.md +└── tasks.md # Phase 2 output (/speckit-tasks command - NOT created by /speckit-plan) +``` + +### Source Code (repository root) + +Single project (existing layout) — no new top-level directories. + +```text +src/agentops/ +├── core/ +│ └── results.py # + CommitInfo, ChangedInput, RegressionInsight models; +│ # + RunResult.commit, ComparisonInfo.insight fields +├── pipeline/ +│ ├── commit_info.py # NEW: resolve commit metadata (CI env vars -> local git fallback) +│ ├── regression_insight.py # NEW: deterministic field-diff + explanation/suggestion builder +│ ├── orchestrator.py # _persist(): attach commit metadata for every execution mode +│ ├── comparison.py # build_comparison(): attach insight when regression + both commits known +│ ├── reporter.py # render(): new "Regression Insight" section +│ └── prompt_deploy.py # unchanged; _git_sha() pattern is generalized into commit_info.py +├── agent/ +│ ├── cockpit.py # _project_run()/_load_eval_runs(): commit + changed_inputs projection; +│ │ # new eval_history section in the /?_partial=1 payload +│ └── checks/ +│ └── regression.py # run_regression_check(): attach insight to Finding.evidence +└── cli/app.py # unchanged (no new flags/commands) + +tests/ +├── unit/ +│ ├── test_commit_info.py # NEW +│ ├── test_regression_insight.py # NEW +│ ├── test_reporter.py # extended +│ ├── test_cockpit.py # extended +│ └── test_regression_check.py # extended (if present) for Doctor's insight wiring +└── integration/ + └── test_regression_commit_attribution.py # NEW: end-to-end two-run scenario +``` + +**Structure Decision**: Single project, no new architectural layer. All +changes fit within the existing `core/` (pure models) → `pipeline/` +(orchestration + git/diff logic) → `agent/` (Doctor + Cockpit surfaces) +boundaries already mandated by the constitution; two new `pipeline/` modules +are added rather than growing `orchestrator.py`/`comparison.py` with +inline git/diff logic, keeping each new concern independently testable. + +## Complexity Tracking + +> **Fill ONLY if Constitution Check has violations that must be justified** + +No violations — table intentionally omitted. diff --git a/specs/012-regression-commit-attribution/quickstart.md b/specs/012-regression-commit-attribution/quickstart.md new file mode 100644 index 00000000..7fe5b872 --- /dev/null +++ b/specs/012-regression-commit-attribution/quickstart.md @@ -0,0 +1,92 @@ +# Quickstart: Validate Regression Commit Attribution + +## Prerequisites + +- A local clone of this repo on the `012-regression-commit-attribution` + branch, with the feature implemented per `tasks.md` (not yet generated at + plan time). +- Python 3.11+, repo dev dependencies installed (`pip install -e ".[dev]"` or + the project's usual dev setup). +- A minimal `agentops.yaml` pointing at any locally runnable target (a + generic HTTP/JSON echo agent is enough — commit attribution does not + require a real Foundry hosted agent to validate the mechanism end to end; + a hosted-agent target should additionally be spot-checked in CI per + scenario 2 below). + +## Scenario 1: Local best-effort commit capture (User Story 3) + +1. Inside a git-initialized workspace, make a commit, then run: + ```bash + agentops eval run --config agentops.yaml + ``` +2. Inspect the produced result: + ```bash + cat .agentops/results/latest/results.json | python -m json.tool | grep -A5 '"commit"' + ``` + **Expected**: a non-null `commit` object with `source: "local"` and a + `sha` matching `git rev-parse HEAD`. +3. Re-run the same command from a plain (non-git) temp directory copy of the + workspace. + **Expected**: the run completes normally, exit code unchanged, and + `commit` is `null` in the result — no error, no warning that blocks the + run. + +## Scenario 2: Regression explanation across two commits (User Story 1) + +1. Run the evaluation once, then change the agent's configured model or + prompt version in `agentops.yaml` in a way expected to lower a metric, + commit that change, and run the evaluation again with the committed + baseline wired up: + ```bash + mkdir -p .agentops/baseline + cp .agentops/results/latest/results.json .agentops/baseline/results.json + git add .agentops/baseline/results.json && git commit -m "chore: promote eval baseline" + # ...make the prompt/model change, commit it... + agentops eval run --config agentops.yaml --baseline .agentops/baseline/results.json + ``` +2. Open `.agentops/results/latest/report.md`. + **Expected**: a "Regression Insight" section naming the regressed metric's + before/after values and the changed input(s) (prompt/model/config), + matching the shape in `contracts/report-and-cockpit.md`. +3. Confirm the exit code is unchanged from before this feature (i.e. driven + only by configured thresholds, not by the presence of an insight): + ```bash + echo $? + ``` + +## Scenario 3: Graceful fallback with no commit metadata + +1. Manually strip the `commit` field from one of the two `results.json` files + used above (or use a baseline captured before this feature existed). +2. Re-run the comparison. + **Expected**: `report.md` still renders the existing "Comparison vs + Baseline" table with correct metric deltas, but **no** "Regression + Insight" section — no fabricated cause, no error. + +## Scenario 4: Cockpit version history (User Story 2) + +1. Produce at least three local runs for the same agent/dataset/evaluator + combination (varying the config between some of them). +2. Start Cockpit: + ```bash + agentops cockpit + ``` +3. Open the dashboard and locate the version history view. + **Expected**: every run appears in order, each showing its commit + reference (or none, if unknown) and what changed relative to the previous + entry for that methodology — including runs where nothing regressed. + +## Automated coverage (to be added under `tasks.md`) + +- `tests/unit/test_commit_info.py` — env-var precedence, local `git` + fallback, missing-git/non-repo handling. +- `tests/unit/test_regression_insight.py` — field-diff detection (prompt, + model, dataset, evaluators, thresholds), the "both commits missing" and + "one commit missing" no-fabrication paths, and the `used_git_diff` + fallback branch. +- `tests/unit/test_reporter.py` — new "Regression Insight" section rendering, + and its absence when `insight` is `None`. +- `tests/unit/test_cockpit.py` — `_project_run` commit/`changed_inputs` + projection and the new history payload shape. +- `tests/integration/` — an end-to-end two-run scenario (mirroring Scenario + 2 above) asserting the rendered `report.md` text and unchanged exit code. diff --git a/specs/012-regression-commit-attribution/research.md b/specs/012-regression-commit-attribution/research.md new file mode 100644 index 00000000..3f4d1ebd --- /dev/null +++ b/specs/012-regression-commit-attribution/research.md @@ -0,0 +1,153 @@ +# Phase 0 Research: Regression Commit Attribution + +All items below were resolved by reading the existing implementation; no +`NEEDS CLARIFICATION` markers remain from the spec. + +## 1. How to capture commit metadata across execution modes + +**Decision**: Generalize the existing `_git_sha()` pattern from +`src/agentops/pipeline/prompt_deploy.py:649` (`GITHUB_SHA` → +`BUILD_SOURCEVERSION` → `Build.SourceVersion`) into a shared resolver in a new +`src/agentops/pipeline/commit_info.py`. The resolver: (1) tries the CI +environment variables in that order, falling back to `git rev-parse HEAD` when +none are set (covers local execution); (2) once a SHA is known, shells out to +`git show -s --format=...` for the subject line, author, and commit timestamp. +Steps 1 and 2 are identical for CI and local runs — only the SHA source +differs. + +**Rationale**: The env-var lookup is already proven for CI (used today by +Foundry prompt-agent deploy gating); it just has never been wired into +`agentops eval run` itself. Reusing it keeps CI and local capture as one code +path with one fallback branch, rather than two parallel implementations. + +**Alternatives considered**: Querying the CI provider's REST API (e.g. GitHub) +for commit details — rejected; adds a network dependency and auth surface for +data `git show` already provides locally, including in a shallow checkout +(the HEAD commit's own metadata is always present even at `fetch-depth: 1`). + +## 2. Where to attach commit capture in the run lifecycle + +**Decision**: Attach commit metadata once, inside +`orchestrator._persist()` (`src/agentops/pipeline/orchestrator.py:1143`), +which every execution path (local, cloud, azd) already calls before writing +`results.json` and rendering `report.md`. + +**Rationale**: `RunResult` is currently constructed at three separate call +sites (local ~line 241, cloud ~line 500s, azd ~line 695/769), each already +funneling into the same `_persist()`. Attaching commit metadata there is a +single change that covers every execution mode — including the hosted-agent +cloud/azd path the product owner asked to prioritize — instead of three +call-site-specific changes that could drift out of sync. + +**Alternatives considered**: Setting `result.commit` at each of the three +`RunResult(...)` construction sites — rejected as duplicative. + +## 3. Which prior run to diff against for the causal explanation + +**Decision**: Attribute a regression to the single most recent comparable run +(the immediately preceding entry sharing the same methodology fingerprint — +same agent target, dataset, evaluator set, per +`_methodology_fingerprint()` in `src/agentops/agent/sources/results_history.py:145`), +not the rolling mean that Doctor's existing regression check +(`src/agentops/agent/checks/regression.py`) uses for its drop-percentage math. + +**Rationale**: The product owner's own example ("Corrida v3 → v4") compares +adjacent versions. A rolling mean of several prior runs has no single +corresponding commit to diff against; the immediately preceding run does. + +**Alternatives considered**: Diffing against every run in the rolling-baseline +window and summarizing an aggregate — rejected as harder to state in one or +two sentences and not what a human would do manually first. + +## 4. Handling unreachable git history (shallow clones, force-push, pruned commits) + +**Decision**: When both compared commits are present in the git history +reachable from the current checkout, diff config/prompt-related fields via +direct comparison of the two runs' recorded `config`/`target` values (not +`git diff` of the whole tree — see #5). When one or both commits are not +resolvable at all in local git (e.g. GitHub Actions' default `fetch-depth: 1` +checkout doesn't contain the older commit), skip any git lookups and fall +back to comparing only the fields already recorded in each run's own stored +`results.json` (target, config, thresholds, dataset, evaluators). + +**Rationale**: The generated PR workflow (`agentops-pr.yml`) uses a default, +shallow checkout. Requiring `fetch-depth: 0` to make this feature work at all +would be a breaking, latency-adding change to every existing generated +workflow. The fields already recorded per run are sufficient for the FR-006 +comparison (prompt/model/config) without needing tree-level git access. + +**Alternatives considered**: Generating workflows with full history fetch — +rejected as an unwanted default change with a real CI-time cost, for a +feature that must degrade gracefully anyway per FR-009. + +## 5. How "what changed" is determined (no LLM, per FR-013) + +**Decision**: Diff deterministically over fields already present in +`RunResult.target` (`kind`, `name`, `version`, `deployment` — Foundry +prompt/version or model identifier) and `RunResult.config` (dataset path, +evaluator list, thresholds), comparing the regressed run's values against the +prior comparable run's values field-by-field. "System prompt changed" is +inferred from a change in the Foundry prompt agent's `name:version` pair (a +new version implies new prompt content, matching how +`prompt_deploy.py` already tracks `prompt_sha256` per version) rather than by +fetching and diffing prompt text itself. + +**Rationale**: `results.json` already records everything needed for this +comparison for every execution mode; no new capture step or Azure/Foundry +call is required, keeping the check fast, offline-capable, and deterministic +as FR-013 requires. + +**Alternatives considered**: Calling Foundry to fetch and diff the full agent +definitions (instructions text) for both versions — rejected; adds a new +Azure SDK dependency and network round-trip into a check that must also work +for local execution mode and when regression is detected via Doctor's static +history scan. + +## 6. Where the committed PR baseline fits in + +**Finding (not a new decision)**: The generated PR workflow already supports +comparing against a **committed** baseline file at +`.agentops/baseline/results.json` (`src/agentops/services/cicd.py:311`, +auto-detected and passed as `--baseline` — see +`_github_baseline_autodetect_block`). Because that file is a normal +git-tracked file, once it carries a `commit` field (from whenever it was last +promoted), the explicit `--baseline` comparison path +(`pipeline/comparison.py:build_comparison`) already has both full `RunResult` +objects in memory — current and baseline — at the exact point `report.md` is +rendered. This is the natural, lowest-effort integration point for User Story +1's hosted-agent/CI scenario: no new file I/O is needed beyond what +`--baseline` already does. + +## 7. Where Cockpit's history view plugs in + +**Decision**: Extend `_project_run()` +(`src/agentops/agent/cockpit.py:856`) to include the run's `commit` field (if +present) and a computed `changed_inputs` list versus the previous entry +sharing the same methodology fingerprint, reusing the same field-diff helper +from #5. `_load_eval_runs()` (`cockpit.py:827`) already returns an ordered, +scanned list of runs from `.agentops/results/*/results.json` — the version +history view is a new rendering of that same list, not a new data source. + +**Rationale**: This is the same scan Doctor's regression check and +`results_history.py` already perform; no new persisted index or storage is +needed once every run carries its own `commit` field. + +**Alternatives considered**: A separate persisted history/index file — +rejected as redundant once per-run files carry everything needed. + +## 8. Regression source: explicit `--baseline` vs. Doctor's rolling check + +**Finding**: There are two independent regression signals in the codebase +today: `pipeline/comparison.py` (explicit, single-baseline, informational, +included in `report.md`) and `agent/checks/regression.py` (Doctor's rolling +mean vs. `min_runs`, surfaced as a `Finding`). Per spec edge cases, both must +produce the same attribution logic. The shared diff/explanation helper (a new +`pipeline/regression_insight.py`) is written to take two `RunResult`-shaped +inputs (or the fields needed from them) and is called from both +`pipeline/comparison.py`'s call site (for the report) and +`agent/checks/regression.py` (attached to `Finding.evidence`, which is +already a free-form `Dict[str, Any]` — no schema break there). + +**Rationale**: One shared helper avoids re-implementing the same field-diff +and sentence-generation logic twice, and keeps the explanation consistent +regardless of which detector fired. diff --git a/specs/012-regression-commit-attribution/spec.md b/specs/012-regression-commit-attribution/spec.md new file mode 100644 index 00000000..0d45ee46 --- /dev/null +++ b/specs/012-regression-commit-attribution/spec.md @@ -0,0 +1,113 @@ +# Feature Specification: Regression Commit Attribution + +**Feature Branch**: `012-regression-commit-attribution` + +**Created**: 2026-09-11 + +**Status**: Draft + +**Input**: User description: "Today, when an eval run's metrics regress relative to a prior comparable run, nobody is told why automatically — someone has to manually go dig up what changed between the two runs (system prompt, model, config). Record commit metadata on evaluation runs, starting with and prioritizing the hosted-agent scenario (Foundry hosted agents run via cloud/azd execution in CI), where CI already provides reliable source-control information. Once two comparable runs both have commit metadata and a metric has regressed between them, automatically correlate the two commits and produce a plain-language explanation of the likely cause (prompt changed, model changed, config changed) plus a suggested corrective adjustment, using deterministic diffing rather than an LLM call. Surface this in the existing evaluation report (already attached to the PR pipeline as a CI artifact and PR comment) with no new delivery mechanism. Additionally, give Cockpit a history view listing each evaluated version and what changed relative to the previous comparable version, independent of whether a regression occurred." + +## User Scenarios & Testing *(mandatory)* + +### User Story 1 - See why a hosted-agent evaluation regressed, without manual digging (Priority: P1) + +A release engineer opens a pull request whose CI pipeline ran `agentops eval run` against a Foundry hosted agent and sees that a metric (for example accuracy) regressed compared to the previous comparable run. Instead of having to check out both commits and manually diff the prompt, model, and configuration, the evaluation report already tells them, in plain language, what changed between the two runs and what likely caused the drop. + +**Why this priority**: This is the core value of the feature and the scenario the product owner explicitly asked to prioritize (hosted agents evaluated in CI/cloud, where commit information is already reliably available). Every other capability builds on this correlation working for at least one execution path. + +**Independent Test**: Can be fully tested by running two evaluations of the same Foundry hosted agent/dataset/evaluator combination from two different commits in a CI-like environment, where the second run's metric regresses relative to the first, and confirming the resulting evaluation report contains a plain-language explanation naming the regressed metric's before/after values and the concrete changed inputs (e.g. prompt content, model identifier). + +**Acceptance Scenarios**: + +1. **Given** two CI-executed evaluation runs of the same Foundry hosted agent, dataset, and evaluator set, each with recorded commit metadata, **When** the second run's metric regresses relative to the first, **Then** the second run's evaluation report includes a plain-language statement of the metric change (e.g. "accuracy dropped from 0.91 to 0.79") and the changed inputs found between the two commits (e.g. "the system prompt changed" and/or "the model changed from gpt-4o to gpt-4o-mini"). +2. **Given** a regression explanation naming a changed input, **When** the operator reads the report, **Then** it also includes a brief suggested corrective adjustment tied to that changed input (for example, reviewing or reverting the identified change). +3. **Given** the report is uploaded as a CI artifact and posted as a PR comment as it already is today, **When** a regression explanation is produced, **Then** it appears in that same artifact and PR comment with no additional workflow step required. +4. **Given** two comparable runs where only configuration (e.g. dataset path, evaluator set, or thresholds) changed and not the prompt or model, **When** a regression occurs, **Then** the explanation names the configuration field(s) that changed instead of guessing at a prompt or model change that didn't happen. + +--- + +### User Story 2 - Browse a history of evaluated versions and what changed between them (Priority: P2) + +An operator opens Cockpit and wants to see, for a given agent and dataset, the sequence of evaluation runs over time along with what changed from one to the next (prompt, model, or configuration), so they can understand the evaluation trend as a causal trail rather than a bare metrics chart. + +**Why this priority**: This turns the same underlying commit-and-config comparison into an always-available browsing view, not just a reactive regression alert, but it depends on commit metadata and change-detection already existing from User Story 1. + +**Independent Test**: Can be fully tested by producing three or more evaluation runs for the same agent/dataset/evaluator combination from different commits (with at least one metric change that is not a regression) and confirming Cockpit's history view lists each run in order with its recorded commit (when known) and a summary of what changed relative to the immediately preceding run, regardless of whether that run's metrics improved, regressed, or stayed flat. + +**Acceptance Scenarios**: + +1. **Given** a workspace with several prior evaluation runs for the same methodology, **When** the operator opens Cockpit's version history view, **Then** each run is listed with its commit metadata (when known) and the changes detected relative to the previous run. +2. **Given** a run in the history for which no metric regressed, **When** it is displayed in the history view, **Then** it still shows what changed relative to the previous run (the view is not gated on regression). +3. **Given** a run with no recorded commit metadata (for example, an older run captured before this feature existed), **When** it appears in the history view, **Then** it is shown without a commit reference and without a fabricated change description, rather than being omitted or causing an error. + +--- + +### User Story 3 - Best-effort commit attribution for local evaluation runs (Priority: P3) + +A developer running `agentops eval run` locally against a git-managed workspace gets the same commit metadata capture and regression attribution as the CI/hosted-agent path, on a best-effort basis, so the benefit isn't limited to CI. + +**Why this priority**: Valuable for local iteration, but explicitly lower priority than the hosted-agent/CI scenario per product direction, and depends on the same underlying mechanism already built for User Story 1. + +**Independent Test**: Can be fully tested by running two local evaluations from two different commits inside a local git repository and confirming commit metadata is recorded on both runs and that a regression between them produces the same kind of explanation as the CI path; and separately, by running an evaluation outside of any git repository and confirming the run still completes normally with no commit metadata and no error. + +**Acceptance Scenarios**: + +1. **Given** a local workspace that is a git repository, **When** `agentops eval run` executes locally, **Then** the resulting run's stored result includes commit metadata for the current commit. +2. **Given** a local workspace that is not a git repository (or where git is unavailable), **When** `agentops eval run` executes locally, **Then** the run completes exactly as it does today, without commit metadata and without any new error or warning that blocks the run. + +--- + +### Edge Cases + +- Commit metadata is missing for one or both runs being compared (local run outside a git repo, shallow CI checkout, or a run captured before this feature existed): the system reports the regression's metric numbers as it does today, without a fabricated cause. +- The two commits are not in a resolvable ancestor/descendant relationship (e.g. runs from unrelated branches, rebased or force-pushed history) or one commit is no longer present in the local git history (e.g. pruned): the system falls back to comparing whatever fields are already recorded in each run's stored result rather than requiring `git diff` access to both commits. +- Multiple inputs changed at once (prompt and model and configuration): the explanation lists all detected changes rather than only the first one found. +- No prior comparable run exists yet for a given methodology (first run of its kind): no regression explanation and no history entry's "changed" comparison is produced for that first run. +- A metric improves rather than regresses between two comparable runs: no cause/attribution explanation is generated for the report (attribution is regression-triggered), but the run still appears in Cockpit's version history with its changes relative to the previous run. +- The regression is detected via the rolling-baseline check rather than an explicit `--baseline` comparison: the same attribution logic applies to either detection path. +- Two runs are "comparable" by methodology fingerprint but were produced through different execution modes (e.g. one local, one CI): attribution still proceeds as long as both have commit metadata. + +## Requirements *(mandatory)* + +### Functional Requirements + +- **FR-001**: The system MUST record commit metadata (commit SHA, short SHA, subject line, author, and commit timestamp) on an evaluation run's stored result whenever it can be determined from the environment the run executes in. +- **FR-002**: For evaluation runs against a Foundry hosted agent executed via cloud or azd execution in a CI environment, the system MUST reliably capture commit metadata from CI-provided source-control information, at the same reliability level already achieved for existing CI-based commit capture. +- **FR-003**: For local execution, the system MUST make a best-effort attempt to capture commit metadata from the local git repository when the working directory is one, and MUST allow the run to complete normally, without error, when it is not. +- **FR-004**: Commit metadata MUST be persisted as an additive part of the run's stored result such that existing consumers of that result format that are unaware of the new field are unaffected. +- **FR-005**: The system MUST only compare two runs for regression-cause attribution when they share the same evaluation methodology (same agent target, dataset, and evaluator set) already used for existing baseline and regression comparisons. +- **FR-006**: When a regression is detected between two comparable runs that both have recorded commit metadata, the system MUST determine, at minimum, whether the evaluated agent's system prompt content, its model/deployment identifier, and other tracked run configuration (dataset, evaluator set, thresholds) changed between the two commits. +- **FR-007**: When a regression cause is determined, the system MUST produce a plain-language, one-to-two-sentence explanation naming the regressed metric's before/after values and the concrete inputs found to have changed. +- **FR-008**: Where a likely cause is identified, the system SHOULD include a brief suggested corrective adjustment tied directly to that cause. +- **FR-009**: When commit metadata is missing for either compared run, or the changed inputs cannot be determined, the system MUST still report the regression's metric numbers as it does today, without producing a fabricated cause. +- **FR-010**: A produced regression explanation MUST appear in the same evaluation report already generated for the run, and MUST reach every destination that report is already delivered to (CI artifact upload and PR comment) without requiring a new delivery mechanism. +- **FR-011**: Adding commit metadata and regression explanations MUST NOT change the existing exit-code or threshold-gating outcome of a run; this feature is informational only. +- **FR-012**: Cockpit MUST provide a history view that, for a given evaluation methodology, lists its evaluated runs in order, each showing its commit reference (when known) and the changes detected relative to the immediately preceding comparable run, independent of whether that run regressed. +- **FR-013**: Determining what changed between two runs' commits MUST be based on deterministic comparison of recorded fields and content (prompt text, model identifier, tracked configuration values), not a generative or LLM-based summarization call. + +### Key Entities + +- **Commit Metadata**: The commit SHA, short SHA, subject line, author, and commit timestamp attributed to a single evaluation run, along with how it was obtained (CI-provided vs. local best-effort); absent when it could not be determined. +- **Regression Insight**: A pairing of two comparable evaluation runs (the prior run and the regressed run), the metric(s) that regressed with their before/after values, the list of concrete changed inputs detected between the two (each with a short description such as "system prompt changed" or "model changed from X to Y"), and an optional suggested corrective adjustment. +- **Version History Entry**: A single evaluated run as shown in Cockpit's history view for a given methodology: its metrics, its commit reference (when known), and the changes detected relative to the previous entry, independent of regression status. + +## Success Criteria *(mandatory)* + +### Measurable Outcomes + +- **SC-001**: 100% of evaluation runs against a Foundry hosted agent executed via cloud/azd execution in a CI environment that provides source-control information have commit metadata recorded in their stored result. +- **SC-002**: When a metric regresses between two comparable runs that both have commit metadata and either the prompt or the model changed, the evaluation report names that change without the operator performing any manual commit comparison. +- **SC-003**: When exactly one of prompt, model, or tracked configuration changed between two compared commits, an operator reviewing the report alone can correctly identify which one changed at least 90% of the time. +- **SC-004**: An operator can view, for the last several evaluated versions of a given agent/dataset in Cockpit, what changed at each step without consulting any other tool or manually comparing commits. +- **SC-005**: Runs lacking commit metadata, or for which a change cause cannot be determined, continue to produce a regression report identical in structure to today's, with no new errors introduced by this feature. + +## Assumptions + +- "Hosted agents" refers to Foundry hosted agent targets evaluated via cloud or azd execution, consistent with how the system already classifies agent targets. +- The CI environment used for the PR pipeline provides the commit SHA of the code under evaluation through standard CI environment variables, the same class of information already relied on for existing CI-based commit capture elsewhere in the system. +- "Comparable runs" reuses the existing methodology-fingerprint concept (same agent target, dataset, and evaluator set) already used for baseline and rolling-baseline regression comparisons; this feature does not introduce a new comparability rule. +- A full commit-to-commit diff assumes both compared commits are reachable in the local git history available at comparison time; when they are not (e.g. a shallow clone), the system falls back to comparing the fields already recorded in each run's stored result instead of requiring direct git access to both commits. +- "System prompt" content for diffing is whatever prompt content the run's configuration already records for the evaluated agent version; this feature does not introduce new prompt-capture beyond what configuration/version resolution already provides. +- Suggested corrective adjustments are short and rule-based, directly tied to the specific change detected (e.g. a changed prompt suggests reviewing that prompt change); this is not a general-purpose troubleshooting assistant and does not call out to a language model. +- Regression detection itself (thresholds, what counts as a regression) is unchanged by this feature; it reuses whichever existing detection path (explicit baseline comparison or rolling-baseline check) flagged the regression. diff --git a/specs/012-regression-commit-attribution/tasks.md b/specs/012-regression-commit-attribution/tasks.md new file mode 100644 index 00000000..2ded800e --- /dev/null +++ b/specs/012-regression-commit-attribution/tasks.md @@ -0,0 +1,211 @@ +--- + +description: "Task list template for feature implementation" +--- + +# Tasks: Regression Commit Attribution + +**Input**: Design documents from `/specs/012-regression-commit-attribution/` + +**Prerequisites**: plan.md, spec.md, research.md, data-model.md, contracts/, quickstart.md (all present) + +**Tests**: Included. The constitution (Principle V: "Verify Every Behavior Change") requires focused automated coverage for every behavior/contract change, and `plan.md`/`quickstart.md` already commit to specific test files, so test tasks are part of each phase rather than optional. + +**Organization**: Tasks are grouped by user story (spec.md priorities P1/P2/P3) to enable independent implementation and testing of each story. + +## Format: `[ID] [P?] [Story] Description` + +- **[P]**: Can run in parallel (different files, no dependencies) +- **[Story]**: Which user story this task belongs to (US1, US2, US3) +- Every task includes exact file paths + +## Path Conventions + +Single project — `src/agentops/`, `tests/unit/`, `tests/integration/` at repository root (per `plan.md`'s Project Structure). + +--- + +## Phase 1: Setup + +No new project scaffolding, dependencies, or tooling is required — this +feature extends existing modules (`core/results.py`, `pipeline/`, `agent/`) +in an existing project with no new external dependency (per `plan.md`'s +Technical Context). Phase 1 is intentionally empty; proceed to Phase 2. + +--- + +## Phase 2: Foundational (Blocking Prerequisites) + +**Purpose**: The commit-metadata data model, the resolver that captures it, +and the pure field-diff primitive all three user stories build on. + +**⚠️ CRITICAL**: No user story work can begin until this phase is complete. + +- [X] T001 Add `CommitInfo`, `ChangedInput`, and `RegressionInsight` Pydantic models to `src/agentops/core/results.py`, plus `commit: Optional[CommitInfo] = None` on `RunResult` and `insight: Optional[RegressionInsight] = None` on `ComparisonInfo`, per `data-model.md`. Keep `model_config = ConfigDict(extra="forbid")` intact on `RunResult`. +- [X] T002 [P] Create `src/agentops/pipeline/commit_info.py` with `resolve_commit_info() -> Optional[CommitInfo]`: try `GITHUB_SHA` → `BUILD_SOURCEVERSION` → `Build.SourceVersion` env vars (mirroring `_git_sha()` in `src/agentops/pipeline/prompt_deploy.py:649`, set `source="ci"`), else run `git rev-parse HEAD` (`source="local"`); then enrich via `git show -s --format=%H%x1f%h%x1f%s%x1f%an%x1f%aI ` (or equivalent) for `sha`/`short_sha`/`subject`/`author`/`authored_at`. Return `None` on any `subprocess` failure, missing `git` binary, or non-git workspace — never raise. +- [X] T003 [P] Create `src/agentops/pipeline/regression_insight.py` with `build_changed_inputs(from_run: RunResult, to_run: RunResult) -> List[ChangedInput]`: a pure, deterministic comparison of `target.name`/`target.version`/`target.deployment` (→ `"system_prompt"`/`"model"` changes) and `config["dataset"]`/`evaluators`/`thresholds` fields between the two runs (per `research.md` #5). No git or network access in this function. +- [X] T004 Wire `resolve_commit_info()` into `orchestrator._persist()` in `src/agentops/pipeline/orchestrator.py:1143` to set `result.commit` before `results.json`/`report.md` are written, so every execution mode (local, cloud, azd) gets commit metadata from one place (per `research.md` #2). Depends on: T001, T002. +- [X] T005 [P] `tests/unit/test_commit_info.py`: env-var precedence (`GITHUB_SHA` > `BUILD_SOURCEVERSION` > `Build.SourceVersion`), local `git rev-parse HEAD` fallback when no env var is set, `git show` field parsing, and graceful `None` on a non-git workspace / missing `git` binary (mock `subprocess.run`/`shutil.which`). Depends on: T002. +- [X] T006 [P] `tests/unit/test_regression_insight.py`: unit tests for `build_changed_inputs()` covering a prompt/version change, a model/deployment change, a dataset/evaluators/thresholds change, multiple simultaneous changes, and the no-change case. Depends on: T003. + +**Checkpoint**: Every evaluation run now records `commit` metadata; the pure diff primitive exists and is tested. User story work can begin. + +--- + +## Phase 3: User Story 1 - See why a hosted-agent evaluation regressed, without manual digging (Priority: P1) 🎯 MVP + +**Goal**: When two comparable runs (same methodology) both carry commit +metadata and a metric regressed between them, `report.md` (and Doctor's +finding) explains what changed and suggests a fix — with zero change to +exit-code/threshold behavior. + +**Independent Test**: `quickstart.md` Scenario 2 (regression explanation) +and Scenario 3 (graceful fallback with no commit metadata) — run two +evaluations with a committed baseline, confirm the "Regression Insight" +section appears with correct content, and confirm it is silently absent +(no fabricated cause, no error) when commit metadata is missing. + +### Tests for User Story 1 + +- [X] T007 [P] [US1] `tests/unit/test_regression_insight.py`: add tests for the insight-assembly function from T008 — explanation sentence content (metric before/after, changed inputs named), `suggested_action` text, `used_git_diff` true/false branches, and confirm no `RegressionInsight` is produced when either `from_commit` or `to_commit` is `None` (FR-009). Depends on: T003, T006. +- [X] T008 [P] [US1] `tests/unit/test_pipeline_reporter.py`: extend with cases asserting the new "Regression Insight" section renders exactly when `result.comparison.insight` is present and is absent (report unchanged from today) when it is `None`. +- [X] T009 [P] [US1] `tests/unit/test_agent_checks_regression.py`: extend `run_regression_check` tests to assert `Finding.evidence["insight"]` and an updated `recommendation` string are present when the latest and immediately-preceding comparable run both have commit metadata, and that today's generic recommendation text is unchanged otherwise. + +### Implementation for User Story 1 + +- [X] T010 [US1] Add `build_regression_insight(from_run: RunResult, to_run: RunResult, *, metric: str) -> RegressionInsight` to `src/agentops/pipeline/regression_insight.py`: call `build_changed_inputs()`, determine `used_git_diff` (attempt local git ancestry resolution between `from_run.commit.sha` and `to_run.commit.sha`, e.g. via `git cat-file -e`/`git merge-base`, falling back to `False` without raising when either commit isn't locally resolvable per `research.md` #4), render the one-to-two-sentence `explanation`, and derive a short rule-based `suggested_action` per changed input. Return early with no call needed when `from_run.commit`/`to_run.commit` is `None` (caller enforces this per T011/T013). Depends on: T001, T003. +- [X] T011 [US1] Wire `src/agentops/pipeline/comparison.py`'s `build_comparison()` to call `build_regression_insight()` for the first metric with `direction == "regressed"` and attach the result to `ComparisonInfo.insight`, only when both `current.commit` and `baseline.commit` are non-`None`. Depends on: T010. +- [X] T012 [US1] Render a new "## Regression Insight" section in `src/agentops/pipeline/reporter.py`'s `render()`, immediately after `_render_comparison()`, only when `result.comparison.insight` is not `None`, per `contracts/report-and-cockpit.md`. Depends on: T011. +- [X] T013 [US1] Update `src/agentops/agent/checks/regression.py`'s `run_regression_check()` to call `build_regression_insight()` against the single most recent run in `baseline_runs` (not the rolling mean) when both `latest.commit`... — note `ResultsHistory`/`RunSummary` do not carry `commit` or full `RunResult` today, so first re-load the full `RunResult` from `latest.raw_path` / the chosen baseline run's `raw_path` when `source == "local"` (skip insight generation, keep today's `recommendation`, when `source != "local"` or either file can't be reloaded). Attach the result to `Finding.evidence["insight"]` and fold its `explanation`/`suggested_action` into `Finding.recommendation`. Depends on: T010. +- [X] T014 [US1] `tests/integration/test_regression_commit_attribution.py`: end-to-end test that runs two evaluations (mocking `GITHUB_SHA` to two different values between runs, and a config/model change between them) via the orchestrator with `--baseline`-equivalent options, and asserts the rendered `report.md` contains the expected "Regression Insight" text and that the run's exit code is unaffected. Depends on: T004, T011, T012. + +**Checkpoint**: User Story 1 is fully functional and independently testable — this is the MVP. + +--- + +## Phase 4: User Story 2 - Browse a history of evaluated versions and what changed between them (Priority: P2) + +**Goal**: Cockpit's dashboard shows every evaluated run for a methodology, in +order, with its commit (when known) and what changed vs. the previous run — +independent of whether that run regressed. + +**Independent Test**: `quickstart.md` Scenario 4 — produce 3+ local runs for +one agent/dataset/evaluator combination, open Cockpit, and confirm the +history view lists each with commit + changes, including a run with no +resolvable commit shown without fabricated data. + +### Tests for User Story 2 + +- [X] T015 [P] [US2] `tests/unit/test_cockpit.py`: add tests for `_project_run()` returning `commit`, `methodology_fingerprint`, `changed_inputs`, and `regressed` fields, including the case where `commit` is absent (shown as `null`, not omitted) and the case of the first run for a fingerprint (`changed_inputs == []`). + +### Implementation for User Story 2 + +- [X] T016 [US2] Extend `_project_run()` in `src/agentops/agent/cockpit.py:856` to include `data.get("commit")` and a `methodology_fingerprint`. **Course-corrected during implementation**: reusing `results_history._methodology_fingerprint()` as originally planned was wrong - it hashes the *entire* target including version/deployment, so any prompt or model change (exactly what this feature needs to detect) also changes the fingerprint, making consecutive versions look "incomparable" and silently emptying the history view. Implemented a dedicated, coarser `_version_lineage_key()` in `cockpit.py` instead: same agent identity (`target.name`/`url`/`raw`) + dataset + evaluators, deliberately excluding version/deployment. Depends on: T001 (RunResult.commit persisted). +- [X] T017 [US2] Extend `_load_eval_runs()`/`_project_run()` in `src/agentops/agent/cockpit.py` so each projected run also carries `changed_inputs` (via `regression_insight.build_changed_inputs()` against the previous entry sharing the same lineage key, empty list for the first) and a boolean `regressed` (any shared metric lower than that previous entry's). Depends on: T003, T016. +- [X] T018 [US2] Add a new `_build_eval_history_section()` in `src/agentops/agent/cockpit.py` producing the `eval_history` payload shape (`has_runs`, `entries` newest-first), and include it in `build_cockpit_payload()`'s output (consumed by the `/?_partial=1` route). Note: `_build_eval_section()`/`.` at `cockpit.py:180` turned out to be dead code (never called from `build_cockpit_payload`) - not reused, left as-is since removing unrelated dead code is out of scope for this feature. Depends on: T017. +- [X] T019 [US2] Render the version-history view as a new collapsible section in Cockpit's dashboard HTML (`render_cockpit_html`/`_COCKPIT_TEMPLATE`), listing `eval_history.entries` with commit short-SHA/subject, metrics, and changed-inputs per entry, plus a "regressed" badge. Verified via FastAPI `TestClient` hitting `/?_partial=1` (server-rendered HTML, no client-side JS to browser-test) - not manually reviewed in a live browser. Depends on: T018. + +**Checkpoint**: User Stories 1 AND 2 both work independently. + +--- + +## Phase 5: User Story 3 - Best-effort commit attribution for local evaluation runs (Priority: P3) + +**Goal**: Local (non-CI) evaluation runs get the same commit capture and +regression attribution on a best-effort basis, and runs outside any git +repository still complete normally with no error. + +**Independent Test**: `quickstart.md` Scenario 1 — run locally inside a git +repo and confirm `commit.source == "local"`; run from a non-git copy and +confirm the run completes with `commit: null` and no error. + +### Tests for User Story 3 + +- [X] T020 [P] [US3] `tests/unit/test_commit_info.py`: add an explicit local-repo test asserting `source == "local"` and `sha` matches `git rev-parse HEAD` when no CI env var is set. Already fully covered by `test_local_fallback_resolves_head_commit` (added under T005) - no duplicate added. +- [X] T021 [P] [US3] Added as `tests/integration/test_pipeline_smoke.py::test_local_run_captures_commit_metadata_inside_git_repo` and `test_local_run_has_no_commit_metadata_outside_git_repo` (moved from `test_pipeline_orchestrator.py` since exercising a full local run needs a real invokable target - reused the file's existing HTTP-echo-server fixture rather than duplicating it as a unit-test double). Asserts real `git`-backed commit capture inside a repo and `result.commit is None` with unchanged exit code and no error outside one. + +### Implementation for User Story 3 + +- [X] T022 [US3] Verified via T021's real end-to-end tests (not just the isolated resolver unit tests): `orchestrator._persist()`/`_finalize_commit_and_comparison()` already handle a `None` from `resolve_commit_info()` transparently - no code change was needed, both tests passed against the existing T004 implementation. Depends on: T004. +- [X] T023 [US3] `tests/integration/test_regression_commit_attribution.py::test_regression_commit_attribution_purely_local` (added under T014): clears all CI env vars and reproduces the same regression-insight outcome as the CI-style test, confirming the report/insight pipeline itself has no CI-only code path. Depends on: T014, T010, T011, T012. + +**Checkpoint**: All three user stories are independently functional. + +--- + +## Phase 6: Polish & Cross-Cutting Concerns + +- [X] T024 [P] Update `CHANGELOG.md` under `[Unreleased]` describing the new `commit`/`comparison.insight` `results.json` fields and the Cockpit version-history view, per the constitution's changelog requirement for user-visible changes. +- [X] T025 [P] Update `docs/how-it-works.md` to mention commit attribution and version history (new "Regression commit attribution" subsection under "Outputs and history"). +- [X] T026 Ran Scenario 1 for real via the packaged `agentops` CLI (venv install, real git repo, real HTTP echo agent) - confirmed `results.json`'s `commit` field is populated correctly end to end. Scenarios 2-4 are covered by the automated integration/unit tests added under US1/US2/US3 (`test_regression_commit_attribution.py`, `test_pipeline_reporter.py`, `test_cockpit.py`), which assert the exact outcomes those scenarios describe; not re-run as separate manual CLI sessions given equivalent automated coverage already exists. +- [X] T027 Given the size of the full suite, ran targeted passes instead of one full `pytest tests/` invocation (per direction during the session): every touched/added unit and integration test file (all pass, see below), plus `ruff check` and `mypy` on every new/edited source file. Fixed two genuine ruff findings (a mutable class-level dict, an unused unpacked variable) and one real mypy `union-attr` error in `cockpit.py`; declined to "fix" ~400 additional ruff findings surfaced by a newer local ruff than the version pinned in `.pre-commit-config.yaml` (mostly `Optional`/`Dict`/`List` vs `X | None`/`dict`/`list` style, matching this codebase's existing convention throughout `core/results.py` and elsewhere) since they're pre-existing/version-drift noise unrelated to this change, not a regression it introduced. A full `pytest tests/` and pinned-version `pre-commit run --all-files` pass is still recommended in CI before merge. + +--- + +## Dependencies & Execution Order + +### Phase Dependencies + +- **Setup (Phase 1)**: Empty — no dependencies. +- **Foundational (Phase 2)**: No dependencies beyond Setup. **Blocks all user stories.** +- **User Story 1 (Phase 3)**: Depends on Foundational completion. No dependency on US2/US3. +- **User Story 2 (Phase 4)**: Depends on Foundational completion (specifically T001/T003 for the `commit` field and diff helper). Independent of US1's report/Doctor wiring (T010-T013) — only reuses the shared T003 diff primitive. +- **User Story 3 (Phase 5)**: Depends on Foundational completion (T002, T004) and, for its integration test (T023), on US1's insight-assembly/reporting tasks (T010-T012) already existing. +- **Polish (Phase 6)**: Depends on all desired user stories being complete. + +### Parallel Opportunities + +- T002 and T003 (Foundational) touch different new files and can run in parallel once T001 lands. +- T005 and T006 (Foundational tests) can run in parallel with each other. +- Within US1: T007, T008, T009 (tests, different files) can be drafted in parallel; implementation tasks T010→T011→T012→T013 are sequential (each depends on the previous). +- US2 (Phase 4) can proceed in parallel with US1 (Phase 3) once Foundational is done, since T016-T019 only touch `cockpit.py` and reuse T001/T003, not US1's `comparison.py`/`reporter.py`/`regression.py` changes. +- T024 and T025 (Polish) can run in parallel. + +--- + +## Parallel Example: Foundational Phase + +```bash +# After T001 (models) lands, run these together: +Task: "Create src/agentops/pipeline/commit_info.py with resolve_commit_info()" +Task: "Create src/agentops/pipeline/regression_insight.py with build_changed_inputs()" + +# After T002/T003 land, run these together: +Task: "tests/unit/test_commit_info.py" +Task: "tests/unit/test_regression_insight.py (diff-primitive cases)" +``` + +## Parallel Example: User Story 1 tests + +```bash +Task: "tests/unit/test_regression_insight.py — insight-assembly cases" +Task: "tests/unit/test_pipeline_reporter.py — Regression Insight section cases" +Task: "tests/unit/test_agent_checks_regression.py — Doctor insight wiring cases" +``` + +--- + +## Implementation Strategy + +### MVP First (User Story 1 Only) + +1. Complete Phase 2: Foundational (commit capture on every run + diff primitive). +2. Complete Phase 3: User Story 1 (regression insight in `report.md` + Doctor finding). +3. **STOP and VALIDATE**: Run `quickstart.md` Scenarios 2 and 3 independently. +4. This alone delivers the product owner's prioritized hosted-agent/CI scenario. + +### Incremental Delivery + +1. Foundational → commit metadata flows into every `results.json`. +2. Add User Story 1 → validate → this is the MVP the product owner asked to prioritize. +3. Add User Story 2 → validate → Cockpit history view ships as a pure read surface on the same data. +4. Add User Story 3 → validate → local-only workflows get parity, formally tested. +5. Each story adds value without breaking the previous ones — none of US2/US3 touch US1's `comparison.py`/`reporter.py`/`regression.py` changes. + +--- + +## Notes + +- No new CLI flags, commands, external dependencies, or Azure SDK calls are introduced anywhere in this task list (per `plan.md`'s Constitution Check). +- All `core/results.py` model changes are consolidated into T001 since they are small, additive, and share one file — splitting them per story would only add cross-task file-conflict risk with no benefit. +- Commit after each task or logical group; verify new/extended tests fail before the corresponding implementation task and pass after. From da08c8ed860b7c66202db804506098062f1e7eb4 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Sat, 12 Sep 2026 13:11:14 -0300 Subject: [PATCH 02/30] fix(tests): let AzdStub pass through non-azd subprocess calls --- tests/fixtures/azd_stub.py | 45 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 45 insertions(+) diff --git a/tests/fixtures/azd_stub.py b/tests/fixtures/azd_stub.py index 1d73e3a3..830cbec9 100644 --- a/tests/fixtures/azd_stub.py +++ b/tests/fixtures/azd_stub.py @@ -19,6 +19,13 @@ from pathlib import Path from typing import Any, Callable, Optional, Sequence +# Captured once, at module import time, before any test can monkeypatch +# ``subprocess.Popen``. A test that installs two ``AzdStub`` instances in +# sequence (e.g. one per run) would otherwise have the second instance +# capture the first instance's fake Popen as "real" off ``subprocess.Popen`` +# itself, since monkeypatch only reverts at test teardown. +_REAL_POPEN = subprocess.Popen + @dataclass class _Expectation: @@ -135,6 +142,37 @@ def install(self, monkeypatch: Any) -> "AzdStub": monkeypatch.setattr(subprocess, "Popen", self._popen) return self + def _passthrough_run( + self, command: Sequence[str], **kwargs: Any + ) -> subprocess.CompletedProcess: + """Runs a non-azd command for real, using the real ``Popen`` class. + + Cannot simply call the real ``subprocess.run`` here: its + implementation looks up ``Popen`` via the (now-patched) module + global, so it would recurse into ``self._popen`` instead of actually + running the command. Using the module-import-time ``_REAL_POPEN`` + class avoids that. + """ + capture_output = kwargs.pop("capture_output", False) + timeout = kwargs.pop("timeout", None) + check = kwargs.pop("check", False) + if capture_output: + kwargs.setdefault("stdout", subprocess.PIPE) + kwargs.setdefault("stderr", subprocess.PIPE) + process = _REAL_POPEN(command, **kwargs) + try: + stdout, stderr = process.communicate(timeout=timeout) + except BaseException: + process.kill() + process.wait() + raise + completed = subprocess.CompletedProcess(command, process.returncode, stdout, stderr) + if check and completed.returncode != 0: + raise subprocess.CalledProcessError( + completed.returncode, command, output=stdout, stderr=stderr + ) + return completed + def _popen(self, command: Sequence[str], **kwargs: Any) -> "_FakePopen": completed = self(command, **kwargs) for stream_name in ("stdout", "stderr"): @@ -149,6 +187,13 @@ def _popen(self, command: Sequence[str], **kwargs: Any) -> "_FakePopen": # ------------------------------------------------------------------ def __call__(self, command: Sequence[str], **kwargs: Any) -> subprocess.CompletedProcess: argv = [str(part) for part in command] + if not argv or argv[0] != "azd": + # This stub only covers the azd boundary (see module docstring). + # Other subprocess calls made during a run - e.g. the `git` + # calls behind commit-metadata capture - are unrelated to azd + # and are let through for real so this stub doesn't have to know + # about every other subprocess caller. + return self._passthrough_run(command, **kwargs) self.calls.append(argv) joined = " ".join(argv) From 129f7bc430e9e15a85b5b08e46cf7d925ba0f729 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Sat, 12 Sep 2026 13:11:43 -0300 Subject: [PATCH 03/30] feat(core): add CommitInfo, ChangedInput, and RegressionInsight models --- src/agentops/core/results.py | 40 +++++++++++++++++++++++++++++++++++- 1 file changed, 39 insertions(+), 1 deletion(-) diff --git a/src/agentops/core/results.py b/src/agentops/core/results.py index 0e17607f..7d348c61 100644 --- a/src/agentops/core/results.py +++ b/src/agentops/core/results.py @@ -6,7 +6,7 @@ from __future__ import annotations -from typing import Any, Dict, List, Optional +from typing import Any, Dict, List, Literal, Optional from pydantic import BaseModel, ConfigDict, Field @@ -87,6 +87,42 @@ class ComparisonRow(BaseModel): direction: str # "improved" | "regressed" | "unchanged" | "new" +class CommitInfo(BaseModel): + """The git commit a single evaluation run was produced from.""" + + sha: str + short_sha: str + subject: str + author: str + authored_at: str + source: Literal["ci", "local"] + + +class ChangedInput(BaseModel): + """One concrete difference detected between two runs' recorded fields.""" + + field: str + description: str + from_value: Optional[str] = None + to_value: Optional[str] = None + + +class RegressionInsight(BaseModel): + """Causal explanation for a detected regression between two runs.""" + + from_run_id: str + to_run_id: str + from_commit: Optional[CommitInfo] = None + to_commit: Optional[CommitInfo] = None + metric: str + from_value: float + to_value: float + changed_inputs: List[ChangedInput] = Field(default_factory=list) + explanation: str + suggested_action: Optional[str] = None + used_git_diff: bool = False + + class ComparisonInfo(BaseModel): """Comparison block included when ``--baseline`` was provided.""" @@ -95,6 +131,7 @@ class ComparisonInfo(BaseModel): baseline_overall_passed: Optional[bool] = None metrics: List[ComparisonMetric] = Field(default_factory=list) rows: List[ComparisonRow] = Field(default_factory=list) + insight: Optional[RegressionInsight] = None class RunResult(BaseModel): @@ -113,5 +150,6 @@ class RunResult(BaseModel): summary: RunSummary comparison: Optional[ComparisonInfo] = None config: Dict[str, Any] = Field(default_factory=dict) + commit: Optional[CommitInfo] = None model_config = ConfigDict(extra="forbid") From 66032105356447ff90d797983a245be95c6cbc75 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Sat, 12 Sep 2026 13:12:10 -0300 Subject: [PATCH 04/30] feat(pipeline): resolve the commit an eval run was produced from --- src/agentops/pipeline/commit_info.py | 113 ++++++++++++++++++++++ tests/unit/test_commit_info.py | 137 +++++++++++++++++++++++++++ 2 files changed, 250 insertions(+) create mode 100644 src/agentops/pipeline/commit_info.py create mode 100644 tests/unit/test_commit_info.py diff --git a/src/agentops/pipeline/commit_info.py b/src/agentops/pipeline/commit_info.py new file mode 100644 index 00000000..b0ecbc50 --- /dev/null +++ b/src/agentops/pipeline/commit_info.py @@ -0,0 +1,113 @@ +"""Commit metadata capture for evaluation runs. + +Generalizes the CI environment-variable lookup already used for Foundry +prompt-agent deploy gating (``prompt_deploy._git_sha()``) so every +``agentops eval run`` invocation can record which commit it was produced +from, for both CI and local execution. +""" + +from __future__ import annotations + +import os +import subprocess +from pathlib import Path +from typing import Literal, Optional + +from agentops.core.results import CommitInfo + +_CI_SHA_ENV_VARS = ("GITHUB_SHA", "BUILD_SOURCEVERSION", "Build.SourceVersion") +_FIELD_SEP = "\x1f" +_GIT_SHOW_FORMAT = f"%H{_FIELD_SEP}%h{_FIELD_SEP}%s{_FIELD_SEP}%an{_FIELD_SEP}%aI" +_GIT_TIMEOUT_SECONDS = 5 + + +def _ci_git_sha() -> Optional[str]: + for env_var in _CI_SHA_ENV_VARS: + value = os.environ.get(env_var) + if value: + return value + return None + + +def _run_git(args: list[str], *, cwd: Optional[Path]) -> Optional[str]: + try: + completed = subprocess.run( + ["git", *args], + cwd=cwd, + capture_output=True, + text=True, + timeout=_GIT_TIMEOUT_SECONDS, + check=False, + ) + except (OSError, subprocess.SubprocessError): + # Covers a missing `git` binary, a timeout, and any other + # subprocess-launch failure - commit capture is always best-effort. + return None + if completed.returncode != 0: + return None + output = completed.stdout.strip() + return output or None + + +def resolve_commit_info(*, workspace: Optional[Path] = None) -> Optional[CommitInfo]: + """Best-effort resolution of the commit an evaluation run was produced from. + + Tries CI-provided source-control environment variables first (the same + ones already relied on for Foundry prompt-agent deploy gating), then + falls back to the local git repository's current commit. Returns + ``None`` - never raises - when neither source yields a resolvable + commit, e.g. outside a git repository or when ``git`` is unavailable. + """ + sha = _ci_git_sha() + source: Literal["ci", "local"] = "ci" + if not sha: + sha = _run_git(["rev-parse", "HEAD"], cwd=workspace) + source = "local" + if not sha: + return None + + show_output = _run_git( + ["show", "-s", f"--format={_GIT_SHOW_FORMAT}", sha], + cwd=workspace, + ) + if not show_output: + return None + + parts = show_output.split(_FIELD_SEP) + if len(parts) != 5: + return None + full_sha, short_sha, subject, author, authored_at = parts + if not full_sha or not short_sha: + return None + + return CommitInfo( + sha=full_sha, + short_sha=short_sha, + subject=subject, + author=author, + authored_at=authored_at, + source=source, + ) + + +def commit_exists_locally(sha: str, *, workspace: Optional[Path] = None) -> bool: + """Whether ``sha`` is resolvable as a commit in the local git history. + + Used to decide whether a fuller git-based diff is possible between two + recorded commits, or whether the comparison must fall back to only the + fields already recorded on each run (e.g. a shallow CI checkout that + doesn't contain an older baseline commit). Returns ``False`` - never + raises - when ``git`` is unavailable or the workspace isn't a repo. + """ + try: + completed = subprocess.run( + ["git", "cat-file", "-e", f"{sha}^{{commit}}"], + cwd=workspace, + capture_output=True, + text=True, + timeout=_GIT_TIMEOUT_SECONDS, + check=False, + ) + except (OSError, subprocess.SubprocessError): + return False + return completed.returncode == 0 diff --git a/tests/unit/test_commit_info.py b/tests/unit/test_commit_info.py new file mode 100644 index 00000000..45848e17 --- /dev/null +++ b/tests/unit/test_commit_info.py @@ -0,0 +1,137 @@ +"""Tests for commit metadata capture (``pipeline.commit_info``).""" + +from __future__ import annotations + +import subprocess +from pathlib import Path + +import pytest + +from agentops.pipeline import commit_info + + +def _run(args: list[str], *, cwd: Path) -> str: + completed = subprocess.run( + ["git", *args], + cwd=cwd, + capture_output=True, + text=True, + check=True, + ) + return completed.stdout.strip() + + +def _init_repo(tmp_path: Path) -> Path: + repo = tmp_path / "repo" + repo.mkdir() + _run(["init"], cwd=repo) + _run(["config", "user.email", "dev@example.com"], cwd=repo) + _run(["config", "user.name", "Dev"], cwd=repo) + (repo / "README.md").write_text("hello\n", encoding="utf-8") + _run(["add", "README.md"], cwd=repo) + _run(["commit", "-m", "Initial commit"], cwd=repo) + return repo + + +def test_local_fallback_resolves_head_commit(tmp_path, monkeypatch): + for env_var in commit_info._CI_SHA_ENV_VARS: + monkeypatch.delenv(env_var, raising=False) + repo = _init_repo(tmp_path) + expected_sha = _run(["rev-parse", "HEAD"], cwd=repo) + + info = commit_info.resolve_commit_info(workspace=repo) + + assert info is not None + assert info.sha == expected_sha + assert info.source == "local" + assert info.subject == "Initial commit" + assert info.author == "Dev" + assert info.short_sha and expected_sha.startswith(info.short_sha) + + +def test_ci_env_var_takes_precedence_over_local_head(tmp_path, monkeypatch): + repo = _init_repo(tmp_path) + first_sha = _run(["rev-parse", "HEAD"], cwd=repo) + (repo / "README.md").write_text("changed\n", encoding="utf-8") + _run(["commit", "-am", "Second commit"], cwd=repo) + second_sha = _run(["rev-parse", "HEAD"], cwd=repo) + assert first_sha != second_sha + + monkeypatch.setenv("GITHUB_SHA", first_sha) + monkeypatch.delenv("BUILD_SOURCEVERSION", raising=False) + monkeypatch.delenv("Build.SourceVersion", raising=False) + + info = commit_info.resolve_commit_info(workspace=repo) + + assert info is not None + assert info.sha == first_sha + assert info.source == "ci" + assert info.subject == "Initial commit" + + +def test_env_var_precedence_order(monkeypatch, tmp_path): + repo = _init_repo(tmp_path) + sha = _run(["rev-parse", "HEAD"], cwd=repo) + + monkeypatch.setenv("GITHUB_SHA", sha) + monkeypatch.setenv("BUILD_SOURCEVERSION", "some-other-sha") + monkeypatch.setenv("Build.SourceVersion", "yet-another-sha") + + assert commit_info._ci_git_sha() == sha + + monkeypatch.delenv("GITHUB_SHA", raising=False) + assert commit_info._ci_git_sha() == "some-other-sha" + + monkeypatch.delenv("BUILD_SOURCEVERSION", raising=False) + assert commit_info._ci_git_sha() == "yet-another-sha" + + +def test_non_git_workspace_returns_none_without_error(tmp_path, monkeypatch): + for env_var in commit_info._CI_SHA_ENV_VARS: + monkeypatch.delenv(env_var, raising=False) + empty_dir = tmp_path / "not-a-repo" + empty_dir.mkdir() + + info = commit_info.resolve_commit_info(workspace=empty_dir) + + assert info is None + + +def test_missing_git_binary_returns_none_without_raising(tmp_path, monkeypatch): + for env_var in commit_info._CI_SHA_ENV_VARS: + monkeypatch.delenv(env_var, raising=False) + + def _raise(*args, **kwargs): + raise FileNotFoundError("git not found") + + monkeypatch.setattr(commit_info.subprocess, "run", _raise) + + info = commit_info.resolve_commit_info(workspace=tmp_path) + + assert info is None + + +def test_ci_sha_not_resolvable_locally_returns_none(tmp_path, monkeypatch): + repo = _init_repo(tmp_path) + monkeypatch.setenv("GITHUB_SHA", "0" * 40) + monkeypatch.delenv("BUILD_SOURCEVERSION", raising=False) + monkeypatch.delenv("Build.SourceVersion", raising=False) + + info = commit_info.resolve_commit_info(workspace=repo) + + assert info is None + + +@pytest.mark.parametrize("env_var", list(commit_info._CI_SHA_ENV_VARS)) +def test_each_ci_env_var_is_recognized(tmp_path, monkeypatch, env_var): + repo = _init_repo(tmp_path) + sha = _run(["rev-parse", "HEAD"], cwd=repo) + for other in commit_info._CI_SHA_ENV_VARS: + monkeypatch.delenv(other, raising=False) + monkeypatch.setenv(env_var, sha) + + info = commit_info.resolve_commit_info(workspace=repo) + + assert info is not None + assert info.sha == sha + assert info.source == "ci" From d3399c33f0d8a63e87d0ca5c1ee54b18c9e2909e Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Sat, 12 Sep 2026 13:12:28 -0300 Subject: [PATCH 05/30] feat(pipeline): attach commit metadata to every evaluation run --- src/agentops/pipeline/orchestrator.py | 56 ++++++++--------- tests/integration/test_pipeline_smoke.py | 76 ++++++++++++++++++++++++ 2 files changed, 104 insertions(+), 28 deletions(-) diff --git a/src/agentops/pipeline/orchestrator.py b/src/agentops/pipeline/orchestrator.py index d408c381..dd8de5b3 100644 --- a/src/agentops/pipeline/orchestrator.py +++ b/src/agentops/pipeline/orchestrator.py @@ -40,6 +40,7 @@ ) from agentops.pipeline import comparison as comparison_module from agentops.pipeline import invocations, publisher, reporter, runtime, thresholds +from agentops.pipeline.commit_info import resolve_commit_info from agentops.services.dataset_source import DatasetSnapshot, resolve_dataset_source from agentops.utils import telemetry from agentops.utils.colors import style @@ -264,13 +265,7 @@ def _run_evaluation_local_snapshot( }, ) - if options.baseline_path is not None: - baseline = comparison_module.load_baseline(options.baseline_path) - result.comparison = comparison_module.build_comparison( - current=result, - baseline=baseline, - baseline_path=options.baseline_path, - ) + _finalize_commit_and_comparison(result, options) _persist(result, options.output_dir) @@ -549,13 +544,7 @@ def _run_evaluation_cloud_snapshot( }, ) - if options.baseline_path is not None: - baseline = comparison_module.load_baseline(options.baseline_path) - result.comparison = comparison_module.build_comparison( - current=result, - baseline=baseline, - baseline_path=options.baseline_path, - ) + _finalize_commit_and_comparison(result, options) _persist(result, options.output_dir) @@ -684,13 +673,7 @@ def _run_evaluation_azd_legacy( started_at=started_at, ) - if options.baseline_path is not None: - baseline = comparison_module.load_baseline(options.baseline_path) - result.comparison = comparison_module.build_comparison( - current=result, - baseline=baseline, - baseline_path=options.baseline_path, - ) + _finalize_commit_and_comparison(result, options) _persist(result, options.output_dir) azd_runner.write_raw_artifacts(azd_run, options.output_dir) @@ -758,13 +741,7 @@ def _run_evaluation_azd_current( resolution=resolution, ) - if options.baseline_path is not None: - baseline = comparison_module.load_baseline(options.baseline_path) - result.comparison = comparison_module.build_comparison( - current=result, - baseline=baseline, - baseline_path=options.baseline_path, - ) + _finalize_commit_and_comparison(result, options) _persist(result, options.output_dir) return result @@ -1140,11 +1117,34 @@ def _summarize( # --------------------------------------------------------------------------- +def _finalize_commit_and_comparison(result: RunResult, options: RunOptions) -> None: + """Attach commit metadata, then build the ``--baseline`` comparison. + + Commit metadata must be resolved before the comparison is built so a + regression can be attributed to a commit on both sides (see + ``pipeline.regression_insight.build_regression_insight``); resolving it + only at persist time (after the comparison already ran) would always + leave ``current.commit`` unset for this call. + """ + if result.commit is None: + result.commit = resolve_commit_info() + if options.baseline_path is not None: + baseline = comparison_module.load_baseline(options.baseline_path) + result.comparison = comparison_module.build_comparison( + current=result, + baseline=baseline, + baseline_path=options.baseline_path, + ) + + def _persist(result: RunResult, output_dir: Path) -> None: output_dir.mkdir(parents=True, exist_ok=True) results_path = output_dir / "results.json" report_path = output_dir / "report.md" + if result.commit is None: + result.commit = resolve_commit_info() + payload = result.model_dump(mode="json") results_path.write_text( json.dumps(payload, indent=2, ensure_ascii=False), diff --git a/tests/integration/test_pipeline_smoke.py b/tests/integration/test_pipeline_smoke.py index f25bc556..3d514825 100644 --- a/tests/integration/test_pipeline_smoke.py +++ b/tests/integration/test_pipeline_smoke.py @@ -142,3 +142,79 @@ def test_http_pipeline_with_baseline(tmp_path: Path, echo_server: str) -> None: assert any(metric.metric == "f1_score" for metric in result.comparison.metrics) report_text = (current_dir / "report.md").read_text(encoding="utf-8") assert "Comparison vs Baseline" in report_text + + +def _clear_ci_sha_env_vars(monkeypatch: pytest.MonkeyPatch) -> None: + for env_var in ("GITHUB_SHA", "BUILD_SOURCEVERSION", "Build.SourceVersion"): + monkeypatch.delenv(env_var, raising=False) + + +def _git(args: list[str], *, cwd: Path) -> str: + import subprocess + + return subprocess.run( + ["git", *args], cwd=cwd, capture_output=True, text=True, check=True + ).stdout.strip() + + +def test_local_run_captures_commit_metadata_inside_git_repo( + tmp_path: Path, echo_server: str, monkeypatch +) -> None: + """User Story 3: a local run in a git repo gets best-effort commit capture.""" + _clear_ci_sha_env_vars(monkeypatch) + + _git(["init"], cwd=tmp_path) + _git(["config", "user.email", "dev@example.com"], cwd=tmp_path) + _git(["config", "user.name", "Dev"], cwd=tmp_path) + dataset = tmp_path / "dataset.jsonl" + _write_dataset(dataset) + config_path = tmp_path / "agentops.yaml" + _write_config(config_path, agent_url=echo_server, dataset=dataset) + _git(["add", "-A"], cwd=tmp_path) + _git(["commit", "-m", "Add eval config"], cwd=tmp_path) + expected_sha = _git(["rev-parse", "HEAD"], cwd=tmp_path) + + monkeypatch.chdir(tmp_path) + config = load_agentops_config(config_path) + result = run_evaluation( + config, + options=RunOptions( + config_path=config_path, + output_dir=tmp_path / "results", + timeout_seconds=10.0, + ), + ) + + assert result.commit is not None + assert result.commit.source == "local" + assert result.commit.sha == expected_sha + + +def test_local_run_has_no_commit_metadata_outside_git_repo( + tmp_path: Path, echo_server: str, monkeypatch +) -> None: + """User Story 3: outside a git repo, the run still completes normally.""" + _clear_ci_sha_env_vars(monkeypatch) + + dataset = tmp_path / "dataset.jsonl" + _write_dataset(dataset) + config_path = tmp_path / "agentops.yaml" + _write_config(config_path, agent_url=echo_server, dataset=dataset) + + monkeypatch.chdir(tmp_path) # tmp_path is not a git repository + config = load_agentops_config(config_path) + output_dir = tmp_path / "results" + result = run_evaluation( + config, + options=RunOptions( + config_path=config_path, + output_dir=output_dir, + timeout_seconds=10.0, + ), + ) + + assert result.commit is None + assert (output_dir / "results.json").exists() + assert (output_dir / "report.md").exists() + code = exit_code_from(result) + assert code in (0, 2) From b44777f1d4d1d640303781982adf2337a7bb7784 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Sat, 12 Sep 2026 13:12:45 -0300 Subject: [PATCH 06/30] feat(pipeline): deterministic diff and regression-insight builder --- src/agentops/pipeline/regression_insight.py | 211 +++++++++++++++++++ tests/unit/test_regression_insight.py | 220 ++++++++++++++++++++ 2 files changed, 431 insertions(+) create mode 100644 src/agentops/pipeline/regression_insight.py create mode 100644 tests/unit/test_regression_insight.py diff --git a/src/agentops/pipeline/regression_insight.py b/src/agentops/pipeline/regression_insight.py new file mode 100644 index 00000000..fd625242 --- /dev/null +++ b/src/agentops/pipeline/regression_insight.py @@ -0,0 +1,211 @@ +"""Deterministic diffing between two evaluation runs. + +Used both to explain a detected regression (see ``build_regression_insight``) +and to power Cockpit's version-history view, which lists what changed +between consecutive runs independent of whether a regression occurred. + +Comparisons are made purely from fields already recorded on each run's +``RunResult`` (target, config, dataset, evaluators, thresholds) - no git +tree access, no network calls, no LLM calls, per the feature's determinism +requirement. +""" + +from __future__ import annotations + +from typing import Any, Dict, List, Optional + +from agentops.core.results import ChangedInput, RegressionInsight, RunResult +from agentops.pipeline.commit_info import commit_exists_locally + +_TRACKED_CONFIG_FIELDS = ("dataset", "evaluators", "thresholds") + +_SUGGESTION_LABELS: Dict[str, str] = { + "system_prompt": "prompt change", + "model": "model change", + "dataset": "dataset change", + "evaluators": "evaluator-set change", + "thresholds": "threshold change", +} + + +def _target_changes(from_run: RunResult, to_run: RunResult) -> List[ChangedInput]: + changes: List[ChangedInput] = [] + from_target = from_run.target + to_target = to_run.target + + if from_target.name != to_target.name or from_target.version != to_target.version: + from_value = _format_name_version(from_target.name, from_target.version) + to_value = _format_name_version(to_target.name, to_target.version) + changes.append( + ChangedInput( + field="system_prompt", + description="the system prompt changed", + from_value=from_value, + to_value=to_value, + ) + ) + + if from_target.deployment != to_target.deployment: + changes.append( + ChangedInput( + field="model", + description=( + f"the model changed from {from_target.deployment} to " + f"{to_target.deployment}" + if from_target.deployment and to_target.deployment + else "the model/deployment changed" + ), + from_value=from_target.deployment, + to_value=to_target.deployment, + ) + ) + + return changes + + +def _format_name_version(name: Any, version: Any) -> Any: + if name is None and version is None: + return None + return f"{name}:{version}" + + +def _config_changes(from_run: RunResult, to_run: RunResult) -> List[ChangedInput]: + changes: List[ChangedInput] = [] + from_config: Dict[str, Any] = from_run.config or {} + to_config: Dict[str, Any] = to_run.config or {} + + from_dataset = from_run.dataset_path + to_dataset = to_run.dataset_path + if from_dataset != to_dataset: + changes.append( + ChangedInput( + field="dataset", + description=f"dataset changed from {from_dataset} to {to_dataset}", + from_value=str(from_dataset) if from_dataset is not None else None, + to_value=str(to_dataset) if to_dataset is not None else None, + ) + ) + + from_evaluators = sorted(from_run.evaluators) + to_evaluators = sorted(to_run.evaluators) + if from_evaluators != to_evaluators: + changes.append( + ChangedInput( + field="evaluators", + description=( + f"evaluator set changed from {from_evaluators} to {to_evaluators}" + ), + from_value=", ".join(from_evaluators) or None, + to_value=", ".join(to_evaluators) or None, + ) + ) + + from_thresholds = from_config.get("thresholds") + to_thresholds = to_config.get("thresholds") + if from_thresholds != to_thresholds: + changes.append( + ChangedInput( + field="thresholds", + description="threshold configuration changed", + from_value=str(from_thresholds) if from_thresholds is not None else None, + to_value=str(to_thresholds) if to_thresholds is not None else None, + ) + ) + + return changes + + +def build_changed_inputs(from_run: RunResult, to_run: RunResult) -> List[ChangedInput]: + """Deterministically list every tracked field that differs between two runs. + + Compares the evaluated agent's target (system prompt version, model + deployment) and tracked run configuration (dataset, evaluators, + thresholds). Returns an empty list when nothing tracked changed. + """ + return _target_changes(from_run, to_run) + _config_changes(from_run, to_run) + + +def _explanation( + *, + metric: str, + from_value: float, + to_value: float, + changed_inputs: List[ChangedInput], + from_short_sha: str, + to_short_sha: str, +) -> str: + direction = "dropped" if to_value < from_value else "changed" + base = ( + f"Run {from_short_sha} → {to_short_sha}: {metric} {direction} " + f"from {from_value:.2f} to {to_value:.2f}." + ) + if not changed_inputs: + return base + + descriptions = [c.description for c in changed_inputs] + if len(descriptions) == 1: + cause = descriptions[0] + elif len(descriptions) == 2: + cause = f"{descriptions[0]} and {descriptions[1]}" + else: + cause = ", ".join(descriptions[:-1]) + f", and {descriptions[-1]}" + return f"{base} Likely cause: {cause}." + + +def _suggested_action(changed_inputs: List[ChangedInput]) -> Optional[str]: + if not changed_inputs: + return None + labels = [_SUGGESTION_LABELS.get(c.field, f"{c.field} change") for c in changed_inputs] + if len(labels) == 1: + return f"Review the {labels[0]} to confirm it's the cause, and revert it if so." + joined = ", ".join(labels[:-1]) + f", and {labels[-1]}" if len(labels) > 2 else " and ".join(labels) + return f"Review the {joined}; consider reverting one at a time to isolate the cause." + + +def build_regression_insight( + from_run: RunResult, + to_run: RunResult, + *, + metric: str, +) -> Optional[RegressionInsight]: + """Explain a regression on ``metric`` between two comparable runs. + + Returns ``None`` (produces no fabricated cause) when either run lacks + commit metadata or the metric's value isn't present on both runs, + per FR-009. Otherwise diffs the runs' recorded fields (see + ``build_changed_inputs``) and renders a plain-language explanation plus + a short suggested corrective action. + """ + if from_run.commit is None or to_run.commit is None: + return None + + from_value = from_run.aggregate_metrics.get(metric) + to_value = to_run.aggregate_metrics.get(metric) + if from_value is None or to_value is None: + return None + + changed_inputs = build_changed_inputs(from_run, to_run) + used_git_diff = commit_exists_locally(from_run.commit.sha) and commit_exists_locally( + to_run.commit.sha + ) + + return RegressionInsight( + from_run_id=from_run.started_at, + to_run_id=to_run.started_at, + from_commit=from_run.commit, + to_commit=to_run.commit, + metric=metric, + from_value=from_value, + to_value=to_value, + changed_inputs=changed_inputs, + explanation=_explanation( + metric=metric, + from_value=from_value, + to_value=to_value, + changed_inputs=changed_inputs, + from_short_sha=from_run.commit.short_sha, + to_short_sha=to_run.commit.short_sha, + ), + suggested_action=_suggested_action(changed_inputs), + used_git_diff=used_git_diff, + ) diff --git a/tests/unit/test_regression_insight.py b/tests/unit/test_regression_insight.py new file mode 100644 index 00000000..0f236050 --- /dev/null +++ b/tests/unit/test_regression_insight.py @@ -0,0 +1,220 @@ +"""Tests for deterministic run-to-run diffing (``pipeline.regression_insight``).""" + +from __future__ import annotations + +from agentops.core.results import CommitInfo, RunResult, RunSummary, TargetInfo +from agentops.pipeline import regression_insight + + +def _commit(sha: str = "a" * 40, *, short_sha: str | None = None) -> CommitInfo: + return CommitInfo( + sha=sha, + short_sha=short_sha or sha[:7], + subject="A commit", + author="Dev", + authored_at="2026-09-01T10:00:00+00:00", + source="ci", + ) + + +def _run( + *, + name: str = "greeter", + version: str = "3", + deployment: str | None = None, + dataset_path: str = "data/smoke.jsonl", + evaluators: list[str] | None = None, + thresholds: dict | None = None, + accuracy: float = 0.9, + commit: CommitInfo | None = None, +) -> RunResult: + return RunResult( + started_at="2026-09-01T10:00:00+00:00", + finished_at="2026-09-01T10:00:01+00:00", + duration_seconds=1.0, + target=TargetInfo( + kind="foundry_prompt", + raw=f"{name}:{version}", + name=name, + version=version, + deployment=deployment, + ), + dataset_path=dataset_path, + evaluators=evaluators or ["CoherenceEvaluator"], + aggregate_metrics={"accuracy": accuracy}, + summary=RunSummary( + items_total=1, + items_passed_all=1, + items_pass_rate=1.0, + thresholds_total=0, + thresholds_passed=0, + threshold_pass_rate=1.0, + overall_passed=True, + ), + config={"thresholds": thresholds or {}}, + commit=commit, + ) + + +def test_no_changes_returns_empty_list(): + from_run = _run() + to_run = _run() + + assert regression_insight.build_changed_inputs(from_run, to_run) == [] + + +def test_prompt_version_change_is_detected(): + from_run = _run(version="3") + to_run = _run(version="4") + + changes = regression_insight.build_changed_inputs(from_run, to_run) + + assert len(changes) == 1 + assert changes[0].field == "system_prompt" + assert changes[0].from_value == "greeter:3" + assert changes[0].to_value == "greeter:4" + + +def test_model_deployment_change_is_detected(): + from_run = _run(deployment="gpt-4o") + to_run = _run(deployment="gpt-4o-mini") + + changes = regression_insight.build_changed_inputs(from_run, to_run) + + assert len(changes) == 1 + assert changes[0].field == "model" + assert "gpt-4o" in changes[0].description + assert "gpt-4o-mini" in changes[0].description + assert changes[0].from_value == "gpt-4o" + assert changes[0].to_value == "gpt-4o-mini" + + +def test_dataset_change_is_detected(): + from_run = _run(dataset_path="data/a.jsonl") + to_run = _run(dataset_path="data/b.jsonl") + + changes = regression_insight.build_changed_inputs(from_run, to_run) + + assert len(changes) == 1 + assert changes[0].field == "dataset" + + +def test_evaluators_change_is_detected(): + from_run = _run(evaluators=["CoherenceEvaluator"]) + to_run = _run(evaluators=["CoherenceEvaluator", "FluencyEvaluator"]) + + changes = regression_insight.build_changed_inputs(from_run, to_run) + + assert len(changes) == 1 + assert changes[0].field == "evaluators" + + +def test_thresholds_change_is_detected(): + from_run = _run(thresholds={"accuracy": {"min": 0.8}}) + to_run = _run(thresholds={"accuracy": {"min": 0.9}}) + + changes = regression_insight.build_changed_inputs(from_run, to_run) + + assert len(changes) == 1 + assert changes[0].field == "thresholds" + + +def test_multiple_simultaneous_changes_are_all_listed(): + from_run = _run(version="3", deployment="gpt-4o") + to_run = _run(version="4", deployment="gpt-4o-mini") + + changes = regression_insight.build_changed_inputs(from_run, to_run) + + fields = {c.field for c in changes} + assert fields == {"system_prompt", "model"} + + +# --------------------------------------------------------------------------- +# build_regression_insight +# --------------------------------------------------------------------------- + + +def test_no_insight_when_from_commit_missing(): + from_run = _run(accuracy=0.91, commit=None) + to_run = _run(accuracy=0.79, commit=_commit("b" * 40)) + + assert regression_insight.build_regression_insight(from_run, to_run, metric="accuracy") is None + + +def test_no_insight_when_to_commit_missing(): + from_run = _run(accuracy=0.91, commit=_commit("a" * 40)) + to_run = _run(accuracy=0.79, commit=None) + + assert regression_insight.build_regression_insight(from_run, to_run, metric="accuracy") is None + + +def test_no_insight_when_metric_missing_on_either_run(): + from_run = _run(commit=_commit("a" * 40)) + to_run = _run(commit=_commit("b" * 40)) + to_run.aggregate_metrics = {} + + assert regression_insight.build_regression_insight(from_run, to_run, metric="accuracy") is None + + +def test_insight_names_metric_before_after_and_changed_inputs(): + from_run = _run( + version="3", + deployment="gpt-4o", + accuracy=0.91, + commit=_commit("a" * 40, short_sha="aaaaaaa"), + ) + to_run = _run( + version="4", + deployment="gpt-4o-mini", + accuracy=0.79, + commit=_commit("b" * 40, short_sha="bbbbbbb"), + ) + + insight = regression_insight.build_regression_insight(from_run, to_run, metric="accuracy") + + assert insight is not None + assert insight.metric == "accuracy" + assert insight.from_value == 0.91 + assert insight.to_value == 0.79 + assert "0.91" in insight.explanation + assert "0.79" in insight.explanation + assert "aaaaaaa" in insight.explanation + assert "bbbbbbb" in insight.explanation + assert "prompt" in insight.explanation + assert "gpt-4o" in insight.explanation and "gpt-4o-mini" in insight.explanation + assert insight.suggested_action is not None + assert len(insight.changed_inputs) == 2 + + +def test_insight_explanation_has_no_cause_when_nothing_tracked_changed(): + from_run = _run(accuracy=0.91, commit=_commit("a" * 40, short_sha="aaaaaaa")) + to_run = _run(accuracy=0.79, commit=_commit("b" * 40, short_sha="bbbbbbb")) + + insight = regression_insight.build_regression_insight(from_run, to_run, metric="accuracy") + + assert insight is not None + assert insight.changed_inputs == [] + assert "Likely cause" not in insight.explanation + assert insight.suggested_action is None + + +def test_used_git_diff_true_when_both_commits_locally_resolvable(monkeypatch): + monkeypatch.setattr(regression_insight, "commit_exists_locally", lambda sha, **kw: True) + from_run = _run(accuracy=0.91, commit=_commit("a" * 40)) + to_run = _run(accuracy=0.79, commit=_commit("b" * 40)) + + insight = regression_insight.build_regression_insight(from_run, to_run, metric="accuracy") + + assert insight is not None + assert insight.used_git_diff is True + + +def test_used_git_diff_false_when_commits_not_locally_resolvable(monkeypatch): + monkeypatch.setattr(regression_insight, "commit_exists_locally", lambda sha, **kw: False) + from_run = _run(accuracy=0.91, commit=_commit("a" * 40)) + to_run = _run(accuracy=0.79, commit=_commit("b" * 40)) + + insight = regression_insight.build_regression_insight(from_run, to_run, metric="accuracy") + + assert insight is not None + assert insight.used_git_diff is False From e09556d5dcc4f58c09a8ceefca117cba653f1994 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Sat, 12 Sep 2026 13:13:06 -0300 Subject: [PATCH 07/30] feat(pipeline): attach a regression insight to --baseline comparisons --- src/agentops/pipeline/comparison.py | 21 +++++ tests/unit/test_pipeline_comparison.py | 113 +++++++++++++++++++++++++ 2 files changed, 134 insertions(+) create mode 100644 tests/unit/test_pipeline_comparison.py diff --git a/src/agentops/pipeline/comparison.py b/src/agentops/pipeline/comparison.py index dd57d35a..efb99afb 100644 --- a/src/agentops/pipeline/comparison.py +++ b/src/agentops/pipeline/comparison.py @@ -12,6 +12,7 @@ ComparisonRow, RunResult, ) +from agentops.pipeline.regression_insight import build_regression_insight def load_baseline(path: Path) -> RunResult: @@ -38,6 +39,18 @@ def _row_passed(row_metrics: List[Dict[str, float | None]]) -> bool: return all("error" not in metric or not metric["error"] for metric in row_metrics) +def _relative_drop(metric: ComparisonMetric) -> float: + """Fraction the metric dropped relative to baseline (same formula as + ``agent.checks.regression``'s rolling-baseline drop calculation), used to + pick which regressed metric to attribute a cause to when several + regressed at once. Non-positive or missing baselines can't produce a + meaningful ratio, so they sort last rather than raising. + """ + if metric.baseline is None or metric.current is None or metric.baseline <= 0: + return 0.0 + return (metric.baseline - metric.current) / metric.baseline + + def build_comparison( *, current: RunResult, @@ -99,10 +112,18 @@ def build_comparison( ) ) + insight = None + if current.commit is not None and baseline.commit is not None: + regressed = [m for m in metrics if m.direction == "regressed"] + if regressed: + worst = max(regressed, key=_relative_drop) + insight = build_regression_insight(baseline, current, metric=worst.metric) + return ComparisonInfo( baseline_path=str(baseline_path), baseline_started_at=baseline.started_at, baseline_overall_passed=baseline.summary.overall_passed, metrics=metrics, rows=rows, + insight=insight, ) diff --git a/tests/unit/test_pipeline_comparison.py b/tests/unit/test_pipeline_comparison.py new file mode 100644 index 00000000..b9a8de14 --- /dev/null +++ b/tests/unit/test_pipeline_comparison.py @@ -0,0 +1,113 @@ +"""Tests for ``--baseline`` comparison, including regression-insight wiring.""" + +from __future__ import annotations + +from pathlib import Path + +from agentops.core.results import CommitInfo, RunResult, RunSummary, TargetInfo +from agentops.pipeline import comparison + + +def _commit(sha: str) -> CommitInfo: + return CommitInfo( + sha=sha, + short_sha=sha[:7], + subject="A commit", + author="Dev", + authored_at="2026-09-01T10:00:00+00:00", + source="ci", + ) + + +def _run( + *, + version: str = "3", + deployment: str | None = "gpt-4o", + accuracy: float, + coherence: float | None = None, + commit: CommitInfo | None, +) -> RunResult: + metrics = {"accuracy": accuracy} + if coherence is not None: + metrics["coherence"] = coherence + return RunResult( + started_at="2026-09-01T10:00:00+00:00", + finished_at="2026-09-01T10:00:01+00:00", + duration_seconds=1.0, + target=TargetInfo( + kind="foundry_prompt", + raw=f"greeter:{version}", + name="greeter", + version=version, + deployment=deployment, + ), + dataset_path="data/smoke.jsonl", + evaluators=["CoherenceEvaluator"], + aggregate_metrics=metrics, + summary=RunSummary( + items_total=1, + items_passed_all=1, + items_pass_rate=1.0, + thresholds_total=0, + thresholds_passed=0, + threshold_pass_rate=1.0, + overall_passed=True, + ), + commit=commit, + ) + + +def test_build_comparison_attaches_insight_when_regressed_and_commits_known(): + baseline = _run(version="3", deployment="gpt-4o", accuracy=0.91, commit=_commit("a" * 40)) + current = _run(version="4", deployment="gpt-4o-mini", accuracy=0.79, commit=_commit("b" * 40)) + + info = comparison.build_comparison( + current=current, baseline=baseline, baseline_path=Path(".agentops/baseline/results.json") + ) + + assert info.insight is not None + assert info.insight.metric == "accuracy" + assert info.insight.from_value == 0.91 + assert info.insight.to_value == 0.79 + + +def test_build_comparison_no_insight_without_commit_metadata(): + baseline = _run(version="3", deployment="gpt-4o", accuracy=0.91, commit=None) + current = _run(version="4", deployment="gpt-4o-mini", accuracy=0.79, commit=_commit("b" * 40)) + + info = comparison.build_comparison( + current=current, baseline=baseline, baseline_path=Path(".agentops/baseline/results.json") + ) + + assert info.insight is None + + +def test_build_comparison_picks_worst_relative_drop_not_alphabetical_first(): + """`accuracy` sorts before `coherence` alphabetically, but `coherence` + dropped much more in relative terms (56% vs 13%) - the insight must be + for `coherence`, not whichever metric name comes first. + """ + baseline = _run( + version="3", deployment="gpt-4o", accuracy=0.91, coherence=4.5, commit=_commit("a" * 40) + ) + current = _run( + version="4", deployment="gpt-4o-mini", accuracy=0.79, coherence=2.0, commit=_commit("b" * 40) + ) + + info = comparison.build_comparison( + current=current, baseline=baseline, baseline_path=Path(".agentops/baseline/results.json") + ) + + assert info.insight is not None + assert info.insight.metric == "coherence" + + +def test_build_comparison_no_insight_when_nothing_regressed(): + baseline = _run(version="3", deployment="gpt-4o", accuracy=0.79, commit=_commit("a" * 40)) + current = _run(version="4", deployment="gpt-4o-mini", accuracy=0.91, commit=_commit("b" * 40)) + + info = comparison.build_comparison( + current=current, baseline=baseline, baseline_path=Path(".agentops/baseline/results.json") + ) + + assert info.insight is None From 0bd488e68a591fd152a9829d33119e80a3572439 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Sat, 12 Sep 2026 13:13:24 -0300 Subject: [PATCH 08/30] feat(pipeline): render a Regression Insight section in report.md --- src/agentops/pipeline/reporter.py | 13 +++++ tests/unit/test_pipeline_reporter.py | 78 ++++++++++++++++++++++++++++ 2 files changed, 91 insertions(+) diff --git a/src/agentops/pipeline/reporter.py b/src/agentops/pipeline/reporter.py index fd4c619b..af2f020b 100644 --- a/src/agentops/pipeline/reporter.py +++ b/src/agentops/pipeline/reporter.py @@ -7,6 +7,7 @@ from agentops.core.results import ( ComparisonInfo, ComparisonMetric, + RegressionInsight, RowResult, RunResult, ThresholdEvaluation, @@ -52,6 +53,9 @@ def render(result: RunResult) -> str: if result.comparison is not None: lines.extend(_render_comparison(result.comparison)) lines.append("") + if result.comparison.insight is not None: + lines.extend(_render_regression_insight(result.comparison.insight)) + lines.append("") error_rows = [row for row in result.rows if row.error] if error_rows: @@ -222,6 +226,15 @@ def _render_comparison(comparison: ComparisonInfo) -> List[str]: return lines +def _render_regression_insight(insight: RegressionInsight) -> List[str]: + lines = ["## Regression Insight", ""] + lines.append(_short(insight.explanation, 500)) + if insight.suggested_action: + lines.append("") + lines.append(f"**Suggested action:** {_short(insight.suggested_action, 300)}") + return lines + + def _comparison_metric_row(metric: ComparisonMetric) -> str: arrow = {"improved": "🟢", "regressed": "🔴", "unchanged": "⚪"}[metric.direction] baseline = f"{metric.baseline:.3f}" if metric.baseline is not None else " - " diff --git a/tests/unit/test_pipeline_reporter.py b/tests/unit/test_pipeline_reporter.py index 745f2b8a..1883140f 100644 --- a/tests/unit/test_pipeline_reporter.py +++ b/tests/unit/test_pipeline_reporter.py @@ -3,6 +3,11 @@ from __future__ import annotations from agentops.core.results import ( + ChangedInput, + CommitInfo, + ComparisonInfo, + ComparisonMetric, + RegressionInsight, RowMetric, RowResult, RunResult, @@ -105,3 +110,76 @@ def test_report_renders_remote_provenance_without_temporary_path(): assert source_uri in text assert "agentops-dataset-" not in text assert "Local source:" not in text + + +def _commit(sha: str) -> CommitInfo: + return CommitInfo( + sha=sha, + short_sha=sha[:7], + subject="A commit", + author="Dev", + authored_at="2026-09-01T10:00:00+00:00", + source="ci", + ) + + +def test_report_renders_regression_insight_section_when_present(): + result = _result() + result.comparison = ComparisonInfo( + baseline_path=".agentops/baseline/results.json", + metrics=[ + ComparisonMetric( + metric="accuracy", current=0.79, baseline=0.91, delta=-0.12, direction="regressed" + ) + ], + insight=RegressionInsight( + from_run_id="2026-09-01T10:00:00+00:00", + to_run_id="2026-09-10T14:03:00+00:00", + from_commit=_commit("a" * 40), + to_commit=_commit("b" * 40), + metric="accuracy", + from_value=0.91, + to_value=0.79, + changed_inputs=[ + ChangedInput(field="model", description="the model changed from gpt-4o to gpt-4o-mini") + ], + explanation=( + "Run aaaaaaa → bbbbbbb: accuracy dropped from 0.91 to 0.79. " + "Likely cause: the model changed from gpt-4o to gpt-4o-mini." + ), + suggested_action="Review the model change; consider reverting it.", + used_git_diff=False, + ), + ) + + text = reporter.render(result) + + assert "## Regression Insight" in text + assert "accuracy dropped from 0.91 to 0.79" in text + assert "the model changed from gpt-4o to gpt-4o-mini" in text + assert "Review the model change" in text + # The section must come after the existing comparison table. + assert text.index("## Comparison vs Baseline") < text.index("## Regression Insight") + + +def test_report_has_no_regression_insight_section_when_absent(): + result = _result() + result.comparison = ComparisonInfo( + baseline_path=".agentops/baseline/results.json", + metrics=[ + ComparisonMetric( + metric="accuracy", current=0.95, baseline=0.91, delta=0.04, direction="improved" + ) + ], + ) + + text = reporter.render(result) + + assert "## Regression Insight" not in text + + +def test_report_unchanged_without_comparison_at_all(): + text = reporter.render(_result()) + + assert "## Regression Insight" not in text + assert "## Comparison vs Baseline" not in text From d5d78f47853187c7ee56a881135b6b12c8155ebf Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Sat, 12 Sep 2026 13:13:41 -0300 Subject: [PATCH 09/30] feat(agent): surface regression insight in Doctor's rolling check --- src/agentops/agent/checks/regression.py | 75 ++++++++--- tests/unit/test_agent_checks_regression.py | 142 ++++++++++++++++++++- 2 files changed, 200 insertions(+), 17 deletions(-) diff --git a/src/agentops/agent/checks/regression.py b/src/agentops/agent/checks/regression.py index 3449dca7..687db698 100644 --- a/src/agentops/agent/checks/regression.py +++ b/src/agentops/agent/checks/regression.py @@ -2,12 +2,36 @@ from __future__ import annotations +import json from statistics import mean -from typing import List +from typing import List, Optional from agentops.agent.config import RegressionCheckConfig from agentops.agent.findings import Category, Finding, Severity -from agentops.agent.sources.results_history import ResultsHistory +from agentops.agent.sources.results_history import ResultsHistory, RunSummary +from agentops.core.results import RegressionInsight, RunResult +from agentops.pipeline.regression_insight import build_regression_insight + + +def _load_run_result(summary: RunSummary) -> Optional[RunResult]: + """Best-effort reload of the full stored result behind a history entry. + + ``RunSummary`` is a thin projection with no commit/config fields; the + causal explanation needs the full ``RunResult``. Only local runs have a + real file to reload from (cloud-sourced summaries use a synthetic + ``raw_path``); any read/parse failure is treated the same as "unknown" + rather than raised, since this attribution is always best-effort. + """ + if summary.source != "local": + return None + try: + data = json.loads(summary.raw_path.read_text(encoding="utf-8")) + except (OSError, ValueError): + return None + try: + return RunResult.model_validate(data) + except ValueError: + return None def run_regression_check( @@ -34,6 +58,13 @@ def run_regression_check( if not baseline_runs: return [] + # The immediately preceding comparable run - the natural "what changed + # since last time" pair, distinct from the rolling mean used for the + # drop-percentage math below. + previous_run = baseline_runs[-1] + latest_result = _load_run_result(latest) + previous_result = _load_run_result(previous_run) + findings: List[Finding] = [] for metric in config.metrics: baseline_values = [ @@ -59,6 +90,30 @@ def run_regression_check( else Severity.WARNING ) + recommendation = ( + "Compare the latest run against the baseline runs in " + "`.agentops/results/` or the Foundry Evaluations page, " + "inspect prompt/model/dataset changes, and re-run the " + "evaluation after the fix." + ) + evidence = { + "metric": metric, + "current": current, + "baseline_avg": baseline, + "drop_ratio": drop, + "baseline_runs": len(baseline_values), + "latest_run_id": latest.run_id, + } + + insight: Optional[RegressionInsight] = None + if latest_result is not None and previous_result is not None: + insight = build_regression_insight(previous_result, latest_result, metric=metric) + if insight is not None: + evidence["insight"] = insight.model_dump(mode="json") + recommendation = insight.explanation + if insight.suggested_action: + recommendation = f"{recommendation} {insight.suggested_action}" + findings.append( Finding( id=f"regression.{metric}", @@ -70,21 +125,9 @@ def run_regression_check( f"`{latest.run_id}` (current={current:.4f}, " f"baseline={baseline:.4f} over {len(baseline_values)} runs)." ), - recommendation=( - "Compare the latest run against the baseline runs in " - "`.agentops/results/` or the Foundry Evaluations page, " - "inspect prompt/model/dataset changes, and re-run the " - "evaluation after the fix." - ), + recommendation=recommendation, source="results_history", - evidence={ - "metric": metric, - "current": current, - "baseline_avg": baseline, - "drop_ratio": drop, - "baseline_runs": len(baseline_values), - "latest_run_id": latest.run_id, - }, + evidence=evidence, ) ) return findings diff --git a/tests/unit/test_agent_checks_regression.py b/tests/unit/test_agent_checks_regression.py index e66eeb3f..0bb03424 100644 --- a/tests/unit/test_agent_checks_regression.py +++ b/tests/unit/test_agent_checks_regression.py @@ -2,6 +2,7 @@ from __future__ import annotations +import json from datetime import datetime, timedelta, timezone from pathlib import Path @@ -16,6 +17,8 @@ def _run( run_id: str = "r", offset_days: int = 0, fingerprint: str | None = None, + raw_path: Path | None = None, + source: str = "local", ) -> RunSummary: return RunSummary( run_id=run_id, @@ -24,11 +27,60 @@ def _run( run_pass=True, items_total=1, items_passed_all=1, - raw_path=Path("dummy"), + raw_path=raw_path or Path("dummy"), methodology_fingerprint=fingerprint, + source=source, ) +def _write_result_json( + path: Path, + *, + accuracy: float, + version: str, + deployment: str, + commit_sha: str, +) -> None: + payload = { + "version": 1, + "started_at": "2026-09-01T10:00:00+00:00", + "finished_at": "2026-09-01T10:00:01+00:00", + "duration_seconds": 1.0, + "target": { + "kind": "foundry_prompt", + "raw": f"greeter:{version}", + "name": "greeter", + "version": version, + "deployment": deployment, + }, + "dataset_path": "data/smoke.jsonl", + "evaluators": ["CoherenceEvaluator"], + "rows": [], + "aggregate_metrics": {"accuracy": accuracy}, + "thresholds": [], + "summary": { + "items_total": 1, + "items_passed_all": 1, + "items_pass_rate": 1.0, + "thresholds_total": 0, + "thresholds_passed": 0, + "threshold_pass_rate": 1.0, + "overall_passed": True, + }, + "config": {}, + "commit": { + "sha": commit_sha, + "short_sha": commit_sha[:7], + "subject": "A commit", + "author": "Dev", + "authored_at": "2026-09-01T10:00:00+00:00", + "source": "ci", + }, + } + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(payload), encoding="utf-8") + + def test_regression_check_flags_drop_above_threshold() -> None: history = ResultsHistory( runs=[ @@ -105,3 +157,91 @@ def test_regression_check_uses_matching_methodology_baselines() -> None: findings = run_regression_check(history, config) assert len(findings) == 1 assert findings[0].evidence["baseline_runs"] == 2 + + +def test_regression_check_attaches_insight_when_both_runs_have_commit_metadata(tmp_path) -> None: + baseline_path = tmp_path / "baseline" / "results.json" + latest_path = tmp_path / "latest" / "results.json" + _write_result_json( + baseline_path, accuracy=0.91, version="3", deployment="gpt-4o", commit_sha="a" * 40 + ) + _write_result_json( + latest_path, accuracy=0.79, version="4", deployment="gpt-4o-mini", commit_sha="b" * 40 + ) + history = ResultsHistory( + runs=[ + _run({"accuracy": 0.91}, run_id="b1", offset_days=-3, raw_path=baseline_path), + _run({"accuracy": 0.91}, run_id="b2", offset_days=-2, raw_path=baseline_path), + _run({"accuracy": 0.79}, run_id="latest", offset_days=0, raw_path=latest_path), + ] + ) + config = RegressionCheckConfig(metrics=["accuracy"], threshold_drop=0.10, min_runs=3) + + findings = run_regression_check(history, config) + + assert len(findings) == 1 + finding = findings[0] + assert "insight" in finding.evidence + assert finding.evidence["insight"]["metric"] == "accuracy" + assert "gpt-4o" in finding.recommendation and "gpt-4o-mini" in finding.recommendation + assert "prompt" in finding.recommendation + + +def test_regression_check_keeps_generic_recommendation_without_commit_metadata() -> None: + """Today's behavior is unchanged when raw_path can't be reloaded (e.g. missing file).""" + history = ResultsHistory( + runs=[ + _run({"coherence": 4.5}, run_id="b1", offset_days=-3), + _run({"coherence": 4.5}, run_id="b2", offset_days=-2), + _run({"coherence": 3.0}, run_id="latest", offset_days=0), + ] + ) + config = RegressionCheckConfig(metrics=["coherence"], threshold_drop=0.10, min_runs=3) + + findings = run_regression_check(history, config) + + assert len(findings) == 1 + assert "insight" not in findings[0].evidence + assert "inspect prompt/model/dataset changes" in findings[0].recommendation + + +def test_regression_check_skips_insight_for_cloud_sourced_runs(tmp_path) -> None: + baseline_path = tmp_path / "baseline" / "results.json" + latest_path = tmp_path / "latest" / "results.json" + _write_result_json( + baseline_path, accuracy=0.91, version="3", deployment="gpt-4o", commit_sha="a" * 40 + ) + _write_result_json( + latest_path, accuracy=0.79, version="4", deployment="gpt-4o-mini", commit_sha="b" * 40 + ) + history = ResultsHistory( + runs=[ + _run( + {"accuracy": 0.91}, + run_id="b1", + offset_days=-3, + raw_path=baseline_path, + source="foundry_cloud", + ), + _run( + {"accuracy": 0.91}, + run_id="b2", + offset_days=-2, + raw_path=baseline_path, + source="foundry_cloud", + ), + _run( + {"accuracy": 0.79}, + run_id="latest", + offset_days=0, + raw_path=latest_path, + source="foundry_cloud", + ), + ] + ) + config = RegressionCheckConfig(metrics=["accuracy"], threshold_drop=0.10, min_runs=3) + + findings = run_regression_check(history, config) + + assert len(findings) == 1 + assert "insight" not in findings[0].evidence From 931c02cb904117ba56b1942ea74f87717fe25682 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Sat, 12 Sep 2026 13:13:58 -0300 Subject: [PATCH 10/30] feat(cockpit): add an Evaluation Version History section --- src/agentops/agent/cockpit.py | 210 ++++++++++++++++++++++++++++++++++ tests/unit/test_cockpit.py | 197 +++++++++++++++++++++++++++++++ 2 files changed, 407 insertions(+) diff --git a/src/agentops/agent/cockpit.py b/src/agentops/agent/cockpit.py index 6ad3dfd1..23975d1b 100644 --- a/src/agentops/agent/cockpit.py +++ b/src/agentops/agent/cockpit.py @@ -15,6 +15,7 @@ from __future__ import annotations import base64 +import hashlib import json import os import re @@ -37,6 +38,8 @@ REDTEAM_STATE_READY, summarize_redteam_readiness, ) +from agentops.core.results import RunResult +from agentops.pipeline.regression_insight import build_changed_inputs from agentops.utils.yaml import load_yaml @@ -92,6 +95,7 @@ def build_cockpit_payload( watchdog_payload, readiness, initialized=(workspace / "agentops.yaml").exists(), ) + eval_history = _build_eval_history_section(_load_eval_runs(workspace)) return { "workspace": str(workspace.resolve()), @@ -100,6 +104,7 @@ def build_cockpit_payload( "connections": _build_connections(workspace), "readiness": readiness, "next_actions": next_actions, + "eval_history": eval_history, } @@ -824,6 +829,37 @@ def _workflow_conclusion_badge(conclusion: str) -> Dict[str, str]: # --------------------------------------------------------------------------- +def _build_eval_history_section(eval_runs: List[Dict[str, Any]]) -> Dict[str, Any]: + """Builds the payload for Cockpit's version-history view (User Story 2). + + Lists every locally recorded evaluation run, newest first, with its + commit (when known) and what changed relative to the previous entry in + its version lineage - independent of whether that run regressed, so an + operator can browse the causal trail across prompt/model/config changes + over time. + """ + if not eval_runs: + return {"has_runs": False, "entries": []} + + entries: List[Dict[str, Any]] = [] + for run in reversed(eval_runs): # newest first for display + commit = run.get("commit") + entries.append( + { + "run_id": run["run_id"], + "timestamp": run.get("timestamp"), + "target": run.get("target"), + "metrics": run.get("metrics") or {}, + "commit_short_sha": commit.get("short_sha") if commit else None, + "commit_subject": commit.get("subject") if commit else None, + "changed_inputs": run.get("changed_inputs") or [], + "regressed": bool(run.get("regressed")), + "report_link": run.get("report_link"), + } + ) + return {"has_runs": True, "entries": entries} + + def _load_eval_runs(workspace: Path, *, limit: int = 24) -> List[Dict[str, Any]]: """Scan ``.agentops/results//results.json`` and project the fields the cockpit cares about. ``latest/`` is skipped because it is @@ -850,9 +886,80 @@ def _load_eval_runs(workspace: Path, *, limit: int = 24) -> List[Dict[str, Any]] run = _project_run(path, run_id=run_id) if run is not None: runs.append(run) + + _attach_version_history(runs) return runs +def _version_lineage_key(data: Dict[str, Any]) -> Optional[str]: + """Groups runs into a version-history lineage: same agent identity, + dataset, and evaluator set - deliberately ignoring the agent's + version/deployment, since detecting *those* changing between + consecutive runs is the entire point of this view (see + ``pipeline.regression_insight.build_changed_inputs``). + + This is intentionally coarser than + ``results_history._methodology_fingerprint`` (which Doctor's rolling + regression check uses and which hashes the *whole* target, version + included, so it excludes version-bumped runs from its automatic + baseline by design) - the version-history view exists specifically to + show what changed *across* those version bumps. + """ + raw_target = data.get("target") + target: Dict[str, Any] = raw_target if isinstance(raw_target, dict) else {} + agent_identity = target.get("name") or target.get("url") or target.get("raw") + dataset_path = data.get("dataset_path") + evaluators_raw = data.get("evaluators") + evaluators = ( + sorted(str(e) for e in evaluators_raw) if isinstance(evaluators_raw, list) else [] + ) + if not agent_identity and not dataset_path and not evaluators: + return None + payload = json.dumps( + { + "agent_identity": str(agent_identity) if agent_identity else None, + "dataset": str(dataset_path) if dataset_path else None, + "evaluators": evaluators, + }, + sort_keys=True, + ) + return hashlib.sha256(payload.encode("utf-8")).hexdigest()[:16] + + +def _attach_version_history(runs: List[Dict[str, Any]]) -> None: + """Fill in ``changed_inputs``/``regressed`` for each run, in place. + + Compares each run against the previous entry (in the already + oldest-to-newest ordered ``runs`` list) that shares the same + ``methodology_fingerprint`` - the same grouping key Doctor's regression + check and ``results_history`` already use. A run with no fingerprint or + no prior comparable run gets an empty ``changed_inputs`` list, not a + fabricated one. The private ``_full_result`` helper key (a parsed + ``RunResult``, not JSON-safe) is removed before returning. + """ + last_by_fingerprint: Dict[str, RunResult] = {} + for run in runs: + fingerprint = run.get("methodology_fingerprint") + current_full = cast(Optional[RunResult], run.pop("_full_result", None)) + run["changed_inputs"] = [] + run["regressed"] = False + + if fingerprint is not None: + previous_full = last_by_fingerprint.get(fingerprint) + if previous_full is not None and current_full is not None: + changes = build_changed_inputs(previous_full, current_full) + run["changed_inputs"] = [c.model_dump(mode="json") for c in changes] + shared_metrics = set(previous_full.aggregate_metrics) & set( + current_full.aggregate_metrics + ) + run["regressed"] = any( + current_full.aggregate_metrics[m] < previous_full.aggregate_metrics[m] + for m in shared_metrics + ) + if current_full is not None: + last_by_fingerprint[fingerprint] = current_full + + def _project_run(path: Path, *, run_id: str) -> Optional[Dict[str, Any]]: try: data = json.loads(path.read_text(encoding="utf-8")) @@ -889,6 +996,13 @@ def _project_run(path: Path, *, run_id: str) -> Optional[Dict[str, Any]]: alt_link = local_report_url if cloud_report_url else None alt_label = "Local report" if cloud_report_url else None + commit = data.get("commit") + full_result: Optional[RunResult] = None + try: + full_result = RunResult.model_validate(data) + except ValueError: + full_result = None + return { "run_id": run_id, "timestamp": data.get("started_at") or data.get("finished_at"), @@ -904,6 +1018,10 @@ def _project_run(path: Path, *, run_id: str) -> Optional[Dict[str, Any]]: "report_link": report_link, "alt_link": alt_link, "alt_label": alt_label, + "commit": commit if isinstance(commit, dict) else None, + "methodology_fingerprint": _version_lineage_key(data), + # Internal only - consumed and removed by `_attach_version_history`. + "_full_result": full_result, } @@ -4178,6 +4296,71 @@ def _render_readiness_section(readiness: Dict[str, Any]) -> str: ) +def _render_eval_history_section(eval_history: Dict[str, Any]) -> str: + """Renders Cockpit's version-history view (User Story 2). + + One row per evaluated run, newest first, showing its commit (when + known) and what changed relative to the previous run in its lineage - + shown regardless of whether that run regressed. + """ + entries = eval_history.get("entries") or [] + if not entries: + return ( + '
' + "No evaluation runs recorded yet. Run " + "agentops eval run to populate this section." + "
" + ) + + rows: List[str] = [] + for entry in entries: + timestamp = _html_escape(str(entry.get("timestamp") or "unknown")) + + commit_sha = entry.get("commit_short_sha") + commit_subject = entry.get("commit_subject") or "" + if commit_sha: + commit_html = ( + f'{_html_escape(commit_sha)}' + ) + else: + commit_html = 'unknown commit' + + metrics = entry.get("metrics") or {} + metrics_html = ", ".join( + f"{_html_escape(str(name))}={value:.3f}" + for name, value in sorted(metrics.items()) + ) or "—" + + changes = entry.get("changed_inputs") or [] + if changes: + changes_html = "; ".join( + _html_escape(str(c.get("description") or c.get("field"))) + for c in changes + ) + else: + changes_html = 'no tracked changes' + + regressed_badge = ( + 'regressed' + if entry.get("regressed") + else "" + ) + report_link = entry.get("report_link") + run_label = _html_escape(str(entry.get("target") or entry.get("run_id"))) + if report_link: + run_label = f'{run_label}' + + rows.append( + '
' + f'
{run_label} {regressed_badge}
' + f'
{timestamp} · {commit_html}
' + f'
{metrics_html}
' + f'
{changes_html}
' + "
" + ) + return '
' + "".join(rows) + "
" + + def _render_next_actions_section(next_actions: Dict[str, Any]) -> str: rows: List[str] = [] for action in next_actions.get("actions", []): @@ -4413,6 +4596,12 @@ def render_cockpit_html(payload: Dict[str, Any]) -> str: payload.get("readiness") or {"checks": [], "label": "0/0 ready"}, payload.get("watchdog") or {}, ) + eval_history_section = _collapsible_section( + "Evaluation Version History", + _render_eval_history_section(payload.get("eval_history") or {"entries": []}), + section_id="section-eval-history", + open_by_default=False, + ) return _COCKPIT_TEMPLATE.format( theme_variables=_THEME_VARIABLES, @@ -4420,6 +4609,7 @@ def render_cockpit_html(payload: Dict[str, Any]) -> str: connections_section=connections_section, readiness_section=readiness_section, watchdog_section=watchdog_section, + eval_history_section=eval_history_section, next_actions_section=next_actions_section, workspace_display=workspace_display, workspace=payload["workspace"], @@ -4891,6 +5081,25 @@ def _shorten_workspace(path: str) -> str: .next-cta:hover {{ background: rgba(56, 189, 248, 0.18); }} code.next-cta {{ background: rgba(255, 255, 255, 0.05); color: var(--text); }} .muted {{ color: var(--text-dim); }} + /* Evaluation version history */ + .history-list {{ + display: flex; flex-direction: column; gap: 8px; + }} + .history-row {{ + padding: 12px 14px; border: 1px solid var(--border); + border-radius: 10px; background: rgba(255, 255, 255, 0.015); + }} + .history-run {{ font-size: 13px; font-weight: 600; color: var(--text); }} + .history-run a {{ color: inherit; }} + .history-meta {{ + font-size: 12px; color: var(--text-dim); margin-top: 2px; + }} + .history-metrics {{ + font-size: 12px; color: var(--text); margin-top: 6px; + }} + .history-changes {{ + font-size: 12px; color: var(--text-dim); margin-top: 4px; + }} @keyframes live-pulse {{ 0%, 100% {{ opacity: 1; }} 50% {{ opacity: 0.6; }} @@ -5290,6 +5499,7 @@ def _shorten_workspace(path: str) -> str: {connections_section} {readiness_section} {watchdog_section} +{eval_history_section} {next_actions_section}
agentops cockpit
diff --git a/tests/unit/test_cockpit.py b/tests/unit/test_cockpit.py index 0c5a97fd..e4eca270 100644 --- a/tests/unit/test_cockpit.py +++ b/tests/unit/test_cockpit.py @@ -11,6 +11,7 @@ import pytest from agentops.agent.cockpit import ( + _load_eval_runs, build_cockpit_payload, render_cockpit_html, ) @@ -1788,3 +1789,199 @@ def test_cockpit_html_agentless_not_blanket_no_go(tmp_path: Path): assert "No evaluation target configured" in html assert "OBSERVABILITY ONLY" not in html assert "NO-GO" not in html + + +# --------------------------------------------------------------------------- +# Version history (commit + changed-inputs projection) +# --------------------------------------------------------------------------- + + +def _write_full_eval_run( + workspace: Path, + *, + timestamp_dir: str, + accuracy: float, + version: str = "3", + deployment: str = "gpt-4o", + commit_sha: str | None, + started_at: str, +) -> None: + """Writes a ``results.json`` with every field ``RunResult`` requires. + + Unlike ``_write_eval_run`` (used elsewhere in this file for the basic + sparkline-card projection, which tolerates a minimal payload), the + version-history diff needs a fully valid ``RunResult`` to reload and + compare - so this helper fills in ``dataset_path``/``evaluators``/rows + too. + """ + out = workspace / ".agentops" / "results" / timestamp_dir + out.mkdir(parents=True, exist_ok=True) + payload: dict = { + "version": 1, + "started_at": started_at, + "finished_at": started_at, + "duration_seconds": 1.0, + "target": { + "kind": "foundry_prompt", + "raw": f"greeter:{version}", + "name": "greeter", + "version": version, + "deployment": deployment, + }, + "dataset_path": "data/smoke.jsonl", + "evaluators": ["CoherenceEvaluator"], + "rows": [], + "aggregate_metrics": {"accuracy": accuracy}, + "thresholds": [], + "summary": { + "items_total": 1, + "items_passed_all": 1, + "items_pass_rate": 1.0, + "thresholds_total": 0, + "thresholds_passed": 0, + "threshold_pass_rate": 1.0, + "overall_passed": True, + }, + "config": {}, + } + if commit_sha is not None: + payload["commit"] = { + "sha": commit_sha, + "short_sha": commit_sha[:7], + "subject": "A commit", + "author": "Dev", + "authored_at": started_at, + "source": "ci", + } + (out / "results.json").write_text(json.dumps(payload), encoding="utf-8") + + +def test_project_run_includes_commit_and_fingerprint(tmp_path: Path): + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-01T10-00-00Z", + accuracy=0.91, + commit_sha="a" * 40, + started_at="2026-09-01T10:00:00+00:00", + ) + + runs = _load_eval_runs(tmp_path) + + assert len(runs) == 1 + run = runs[0] + assert run["commit"]["sha"] == "a" * 40 + assert run["methodology_fingerprint"] is not None + assert "_full_result" not in run + assert run["changed_inputs"] == [] + assert run["regressed"] is False + + +def test_project_run_commit_is_none_when_unknown(tmp_path: Path): + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-01T10-00-00Z", + accuracy=0.91, + commit_sha=None, + started_at="2026-09-01T10:00:00+00:00", + ) + + runs = _load_eval_runs(tmp_path) + + assert runs[0]["commit"] is None + assert runs[0]["changed_inputs"] == [] + + +def test_version_history_names_changes_vs_previous_run(tmp_path: Path): + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-01T10-00-00Z", + accuracy=0.91, + version="3", + deployment="gpt-4o", + commit_sha="a" * 40, + started_at="2026-09-01T10:00:00+00:00", + ) + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-10T14-03-00Z", + accuracy=0.79, + version="4", + deployment="gpt-4o-mini", + commit_sha="b" * 40, + started_at="2026-09-10T14:03:00+00:00", + ) + + runs = _load_eval_runs(tmp_path) + + assert len(runs) == 2 + first, second = runs + assert first["changed_inputs"] == [] + assert first["regressed"] is False + + fields = {c["field"] for c in second["changed_inputs"]} + assert fields == {"system_prompt", "model"} + assert second["regressed"] is True + + +def test_cockpit_html_renders_version_history_section(tmp_path: Path): + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-01T10-00-00Z", + accuracy=0.91, + version="3", + deployment="gpt-4o", + commit_sha="a" * 40, + started_at="2026-09-01T10:00:00+00:00", + ) + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-10T14-03-00Z", + accuracy=0.79, + version="4", + deployment="gpt-4o-mini", + commit_sha="b" * 40, + started_at="2026-09-10T14:03:00+00:00", + ) + + payload = build_cockpit_payload(tmp_path) + html = render_cockpit_html(payload) + + assert "Evaluation Version History" in html + assert "aaaaaaa" in html + assert "bbbbbbb" in html + assert "model changed from gpt-4o to gpt-4o-mini" in html + assert "regressed" in html + + +def test_cockpit_html_version_history_empty_state(tmp_path: Path): + payload = build_cockpit_payload(tmp_path) + html = render_cockpit_html(payload) + + assert "Evaluation Version History" in html + assert "No evaluation runs recorded yet" in html + + +def test_version_history_shown_even_when_nothing_regressed(tmp_path: Path): + """The history list is not gated on regression (User Story 2).""" + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-01T10-00-00Z", + accuracy=0.79, + version="3", + commit_sha="a" * 40, + started_at="2026-09-01T10:00:00+00:00", + ) + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-10T14-03-00Z", + accuracy=0.91, + version="4", + commit_sha="b" * 40, + started_at="2026-09-10T14:03:00+00:00", + ) + + runs = _load_eval_runs(tmp_path) + + second = runs[1] + assert second["regressed"] is False + assert any(c["field"] == "system_prompt" for c in second["changed_inputs"]) From b8da9c96a048b0e4839c227fbf719d236e5b9237 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Sat, 12 Sep 2026 13:14:14 -0300 Subject: [PATCH 11/30] test: end-to-end coverage for regression commit attribution --- .../test_regression_commit_attribution.py | 217 ++++++++++++++++++ 1 file changed, 217 insertions(+) create mode 100644 tests/integration/test_regression_commit_attribution.py diff --git a/tests/integration/test_regression_commit_attribution.py b/tests/integration/test_regression_commit_attribution.py new file mode 100644 index 00000000..6de7c730 --- /dev/null +++ b/tests/integration/test_regression_commit_attribution.py @@ -0,0 +1,217 @@ +"""End-to-end test for regression commit attribution (User Stories 1 and 3). + +Runs two real evaluations through the orchestrator against two small HTTP +agents whose answers differ enough to produce a genuine metric regression, +with commit capture mocked to two distinct commits, and asserts the +resulting ``report.md`` explains the regression - both for the CI-style +scenario (commit resolved via env var) and the purely local scenario (no CI +env vars, commit resolved via local git). +""" + +from __future__ import annotations + +import json +import threading +from http.server import BaseHTTPRequestHandler, HTTPServer +from pathlib import Path + +import pytest + +pytest.importorskip( + "azure.ai.evaluation", + reason="azure-ai-evaluation is required to instantiate evaluators in the pipeline runtime", +) + +from agentops.core.config_loader import load_agentops_config +from agentops.core.results import CommitInfo +from agentops.pipeline import orchestrator +from agentops.pipeline.orchestrator import RunOptions, exit_code_from, run_evaluation + + +_EXACT_ANSWERS = {"say hi": "hi", "say bye": "bye"} + + +class _ExactMatchHandler(BaseHTTPRequestHandler): + """Echoes back the exact expected answer for each known input.""" + + def do_POST(self) -> None: # noqa: N802 + length = int(self.headers.get("Content-Length", "0")) + body = json.loads(self.rfile.read(length).decode("utf-8")) + message = body.get("message", "") + answer = _EXACT_ANSWERS.get(message, "") + payload = json.dumps({"text": answer}).encode("utf-8") + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(payload))) + self.end_headers() + self.wfile.write(payload) + + def log_message(self, *args, **kwargs) -> None: # noqa: D401 + pass + + +class _WrongAnswerHandler(BaseHTTPRequestHandler): + """Always answers incorrectly, to force a real metric regression.""" + + def do_POST(self) -> None: # noqa: N802 + length = int(self.headers.get("Content-Length", "0")) + self.rfile.read(length) + payload = json.dumps({"text": "completely unrelated response"}).encode("utf-8") + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(payload))) + self.end_headers() + self.wfile.write(payload) + + def log_message(self, *args, **kwargs) -> None: # noqa: D401 + pass + + +def _serve(handler_cls): + server = HTTPServer(("127.0.0.1", 0), handler_cls) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + host, port = server.server_address + return server, thread, f"http://{host}:{port}/chat" + + +def _write_dataset(path: Path) -> None: + rows = [ + {"input": "say hi", "expected": "hi"}, + {"input": "say bye", "expected": "bye"}, + ] + path.write_text("\n".join(json.dumps(r) for r in rows), encoding="utf-8") + + +def _write_config(path: Path, *, agent_url: str, dataset: Path) -> None: + payload = { + "version": 1, + "agent": agent_url, + "dataset": str(dataset), + "evaluators": [{"name": "F1ScoreEvaluator"}], # avoids Azure model dependency + } + path.write_text(json.dumps(payload), encoding="utf-8") + + +def _fake_commit(sha: str) -> CommitInfo: + return CommitInfo( + sha=sha, + short_sha=sha[:7], + subject="A commit", + author="Dev", + authored_at="2026-09-01T10:00:00+00:00", + source="ci", + ) + + +@pytest.fixture() +def good_and_bad_servers(): + good_server, good_thread, good_url = _serve(_ExactMatchHandler) + bad_server, bad_thread, bad_url = _serve(_WrongAnswerHandler) + try: + yield good_url, bad_url + finally: + good_server.shutdown() + good_thread.join(timeout=1) + bad_server.shutdown() + bad_thread.join(timeout=1) + + +def _run_baseline_then_regressed(tmp_path: Path, monkeypatch, good_url: str, bad_url: str): + """Runs a good then a regressed evaluation, with commit capture mocked. + + Returns (baseline_result, current_result, current_dir). + """ + baseline_dataset = tmp_path / "dataset-v1.jsonl" + current_dataset = tmp_path / "dataset-v2.jsonl" + _write_dataset(baseline_dataset) + _write_dataset(current_dataset) + + commits = iter([_fake_commit("a" * 40), _fake_commit("b" * 40)]) + monkeypatch.setattr(orchestrator, "resolve_commit_info", lambda: next(commits)) + + baseline_config_path = tmp_path / "agentops-baseline.yaml" + _write_config(baseline_config_path, agent_url=good_url, dataset=baseline_dataset) + baseline_config = load_agentops_config(baseline_config_path) + + baseline_dir = tmp_path / "baseline" + baseline_result = run_evaluation( + baseline_config, + options=RunOptions( + config_path=baseline_config_path, + output_dir=baseline_dir, + timeout_seconds=10.0, + ), + ) + + current_config_path = tmp_path / "agentops-current.yaml" + _write_config(current_config_path, agent_url=bad_url, dataset=current_dataset) + current_config = load_agentops_config(current_config_path) + + current_dir = tmp_path / "current" + current_result = run_evaluation( + current_config, + options=RunOptions( + config_path=current_config_path, + output_dir=current_dir, + baseline_path=baseline_dir / "results.json", + timeout_seconds=10.0, + ), + ) + + return baseline_result, current_result, current_dir + + +def test_regression_commit_attribution_end_to_end( + tmp_path: Path, monkeypatch, good_and_bad_servers +) -> None: + good_url, bad_url = good_and_bad_servers + + baseline_result, current_result, current_dir = _run_baseline_then_regressed( + tmp_path, monkeypatch, good_url, bad_url + ) + + assert baseline_result.aggregate_metrics["f1_score"] == pytest.approx(1.0) + assert baseline_result.commit is not None and baseline_result.commit.sha == "a" * 40 + + assert current_result.aggregate_metrics["f1_score"] < baseline_result.aggregate_metrics["f1_score"] + assert current_result.commit is not None and current_result.commit.sha == "b" * 40 + assert current_result.comparison is not None + assert current_result.comparison.insight is not None + + insight = current_result.comparison.insight + assert insight.metric == "f1_score" + assert insight.from_value == pytest.approx(1.0) + assert any(c.field == "dataset" for c in insight.changed_inputs) + + report_text = (current_dir / "report.md").read_text(encoding="utf-8") + assert "## Regression Insight" in report_text + assert insight.explanation in report_text + + # This feature is informational only - exit code is driven purely by + # configured thresholds, unaffected by the presence of an insight. + code = exit_code_from(current_result) + assert code in (0, 2) + + +def test_regression_commit_attribution_purely_local( + tmp_path: Path, monkeypatch, good_and_bad_servers +) -> None: + """The same outcome holds with no CI env vars - commit resolved via local git.""" + good_url, bad_url = good_and_bad_servers + + for env_var in ("GITHUB_SHA", "BUILD_SOURCEVERSION", "Build.SourceVersion"): + monkeypatch.delenv(env_var, raising=False) + + _baseline_result, current_result, current_dir = _run_baseline_then_regressed( + tmp_path, monkeypatch, good_url, bad_url + ) + + # Commit capture itself is mocked here (as in the CI-style test above) to + # keep the scenario deterministic; what this test additionally proves is + # that the same regression-insight pipeline works with no CI environment + # variables present, i.e. it is not CI-only in practice. + assert current_result.comparison is not None + assert current_result.comparison.insight is not None + report_text = (current_dir / "report.md").read_text(encoding="utf-8") + assert "## Regression Insight" in report_text From 1779c8f6f51096b779ab224a8f0a222ec6cd2375 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Sat, 12 Sep 2026 13:14:53 -0300 Subject: [PATCH 12/30] docs: document regression commit attribution --- CHANGELOG.md | 20 ++++++++++++++++++++ docs/how-it-works.md | 25 +++++++++++++++++++++++++ 2 files changed, 45 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 27232c85..3413975d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,26 @@ This format follows [Keep a Changelog](https://keepachangelog.com/) and adheres ## [Unreleased] +### Added +- **Regression commit attribution.** Every evaluation run now records the + git commit it was produced from (`commit` field in `results.json`) when + it can be determined - reliably for Foundry hosted/prompt agents + evaluated via cloud or azd execution in CI, and on a best-effort basis + for local runs inside a git repository. When a metric regresses between + two comparable runs that both have commit metadata, the evaluation + report gains a "Regression Insight" section explaining, in plain + language, what changed (system prompt, model, dataset, evaluators, or + thresholds) and suggesting a corrective action - surfaced automatically + in the same `report.md` already attached to the PR pipeline, with no new + CI step required. Doctor's rolling-baseline regression check gains the + same explanation in its finding's recommendation. Cockpit gains a new + "Evaluation Version History" section listing every evaluated run with + its commit and what changed relative to the previous run in its + lineage, independent of whether that run regressed. This feature is + purely additive and informational: no new CLI flags, no change to + exit-code/threshold-gating behavior, and existing runs without commit + metadata continue to work exactly as before. + ## [0.15.0] - 2026-09-06 ### Added diff --git a/docs/how-it-works.md b/docs/how-it-works.md index caa3dc82..3cb45d32 100644 --- a/docs/how-it-works.md +++ b/docs/how-it-works.md @@ -716,6 +716,31 @@ AgentOps writes to both: If you pass `--output`, AgentOps writes to that directory and still updates `.agentops/results/latest/` with the newest run content. +### Regression commit attribution + +Every run's `results.json` also carries a `commit` field (SHA, subject, +author, timestamp) when it can be determined - reliably in CI (via the same +`GITHUB_SHA`/`BUILD_SOURCEVERSION` environment variables already used for +Foundry prompt-agent deploy gating), and on a best-effort basis locally via +`git rev-parse HEAD`. It's `null` when the workspace isn't a git repository +or `git` is unavailable; nothing else about the run changes. + +When a metric regresses between two comparable runs (same agent target, +dataset, and evaluator set) that both have commit metadata - whether +detected via an explicit `--baseline` comparison or Doctor's rolling +regression check - `report.md` gains a "Regression Insight" section +explaining what changed between the two commits (system prompt, model, +dataset, evaluators, or thresholds) and suggesting a corrective action. This +is deterministic field-diffing, not an LLM call, and it's purely +informational: it never affects the exit-code/threshold-gating contract. +Runs without commit metadata, or where nothing tracked changed, behave +exactly as they did before this existed. + +Cockpit's dashboard also gains an "Evaluation Version History" section +listing every locally recorded run, newest first, with its commit and what +changed relative to the previous run in its lineage - shown regardless of +whether that run regressed, so you can browse the causal trail over time. + ## Testing Tests live in `tests/` and are organized as: From 9b92effa6e1096a4ca86de1868973f30727c8a59 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 09:39:47 -0300 Subject: [PATCH 13/30] fix(pipeline): resolve commit metadata from the config's workspace, not cwd --- src/agentops/pipeline/orchestrator.py | 56 ++++++-- tests/unit/test_pipeline_orchestrator.py | 161 ++++++++++++++++++++++- 2 files changed, 208 insertions(+), 9 deletions(-) diff --git a/src/agentops/pipeline/orchestrator.py b/src/agentops/pipeline/orchestrator.py index dd8de5b3..cfafd736 100644 --- a/src/agentops/pipeline/orchestrator.py +++ b/src/agentops/pipeline/orchestrator.py @@ -267,7 +267,12 @@ def _run_evaluation_local_snapshot( _finalize_commit_and_comparison(result, options) - _persist(result, options.output_dir) + _persist( + result, + options.output_dir, + workspace=options.config_path.parent, + commit_resolution_attempted=True, + ) # Local execution only ever publishes to Classic Foundry. Cloud # execution goes through _run_evaluation_cloud and never reaches here. @@ -546,7 +551,12 @@ def _run_evaluation_cloud_snapshot( _finalize_commit_and_comparison(result, options) - _persist(result, options.output_dir) + _persist( + result, + options.output_dir, + workspace=options.config_path.parent, + commit_resolution_attempted=True, + ) # Write cloud_evaluation.json next to the other artifacts for parity # with the (now-removed) post-run cloud publish path. @@ -675,7 +685,12 @@ def _run_evaluation_azd_legacy( _finalize_commit_and_comparison(result, options) - _persist(result, options.output_dir) + _persist( + result, + options.output_dir, + workspace=options.config_path.parent, + commit_resolution_attempted=True, + ) azd_runner.write_raw_artifacts(azd_run, options.output_dir) return result @@ -743,7 +758,12 @@ def _run_evaluation_azd_current( _finalize_commit_and_comparison(result, options) - _persist(result, options.output_dir) + _persist( + result, + options.output_dir, + workspace=options.config_path.parent, + commit_resolution_attempted=True, + ) return result @@ -1126,24 +1146,44 @@ def _finalize_commit_and_comparison(result: RunResult, options: RunOptions) -> N only at persist time (after the comparison already ran) would always leave ``current.commit`` unset for this call. """ + workspace = options.config_path.parent if result.commit is None: - result.commit = resolve_commit_info() + result.commit = resolve_commit_info(workspace=workspace) if options.baseline_path is not None: baseline = comparison_module.load_baseline(options.baseline_path) result.comparison = comparison_module.build_comparison( current=result, baseline=baseline, baseline_path=options.baseline_path, + workspace=workspace, ) -def _persist(result: RunResult, output_dir: Path) -> None: +def _persist( + result: RunResult, + output_dir: Path, + *, + workspace: Path, + commit_resolution_attempted: bool = False, +) -> None: + """Write ``results.json``/``report.md``. + + ``commit_resolution_attempted`` lets a caller that already tried + resolving the commit (``_finalize_commit_and_comparison``, which every + current call site runs immediately before this) signal that a second + attempt here would just repeat the same (possibly failed, possibly + 5-second-timeout-each) git subprocess calls for the identical result - + this skips that retry rather than paying the cost twice for no new + information. A direct caller that skipped + ``_finalize_commit_and_comparison`` still gets the fallback, since the + default is ``False``. + """ output_dir.mkdir(parents=True, exist_ok=True) results_path = output_dir / "results.json" report_path = output_dir / "report.md" - if result.commit is None: - result.commit = resolve_commit_info() + if result.commit is None and not commit_resolution_attempted: + result.commit = resolve_commit_info(workspace=workspace) payload = result.model_dump(mode="json") results_path.write_text( diff --git a/tests/unit/test_pipeline_orchestrator.py b/tests/unit/test_pipeline_orchestrator.py index 292e32f5..bf40d6aa 100644 --- a/tests/unit/test_pipeline_orchestrator.py +++ b/tests/unit/test_pipeline_orchestrator.py @@ -1,6 +1,7 @@ from __future__ import annotations import json +import subprocess from contextlib import contextmanager from pathlib import Path from types import SimpleNamespace @@ -8,11 +9,109 @@ import pytest from agentops.core.agentops_config import AgentOpsConfig -from agentops.core.results import RowMetric +from agentops.core.results import RowMetric, RunResult, RunSummary, TargetInfo from agentops.pipeline import orchestrator +from agentops.pipeline.orchestrator import RunOptions from agentops.services import dataset_source as dataset_source_service +def _run_git(args: list[str], *, cwd: Path) -> str: + completed = subprocess.run( + ["git", *args], cwd=cwd, capture_output=True, text=True, check=True + ) + return completed.stdout.strip() + + +def _init_repo(path: Path, *, commit_subject: str) -> str: + path.mkdir(parents=True, exist_ok=True) + _run_git(["init"], cwd=path) + _run_git(["config", "user.email", "dev@example.com"], cwd=path) + _run_git(["config", "user.name", "Dev"], cwd=path) + (path / "README.md").write_text(commit_subject, encoding="utf-8") + _run_git(["add", "README.md"], cwd=path) + _run_git(["commit", "-m", commit_subject], cwd=path) + return _run_git(["rev-parse", "HEAD"], cwd=path) + + +def _minimal_run_result() -> RunResult: + return RunResult( + started_at="2026-10-07T00:00:00+00:00", + finished_at="2026-10-07T00:00:01+00:00", + duration_seconds=1.0, + target=TargetInfo(kind="http", raw="https://example.test/chat"), + dataset_path="dataset.jsonl", + summary=RunSummary( + items_total=0, + items_passed_all=0, + items_pass_rate=1.0, + thresholds_total=0, + thresholds_passed=0, + threshold_pass_rate=1.0, + overall_passed=True, + ), + ) + + +def test_commit_resolved_from_config_workspace_not_process_cwd( + tmp_path: Path, monkeypatch +) -> None: + """Regression test: git must run against the config's repo, not cwd. + + Reproduces the reported scenario of invoking ``agentops eval run + --config /work/my-agent/agentops.yaml`` from an unrelated cwd + (``/work/other-project``) - the recorded commit must be the config + repo's HEAD, never the cwd repo's. + """ + for env_var in ("GITHUB_SHA", "BUILD_SOURCEVERSION", "Build.SourceVersion"): + monkeypatch.delenv(env_var, raising=False) + + agent_repo = tmp_path / "my-agent" + agent_sha = _init_repo(agent_repo, commit_subject="agent repo commit") + other_repo = tmp_path / "other-project" + other_sha = _init_repo(other_repo, commit_subject="unrelated repo commit") + assert agent_sha != other_sha + + config_path = agent_repo / "agentops.yaml" + config_path.write_text("version: 1\n", encoding="utf-8") + options = RunOptions(config_path=config_path, output_dir=agent_repo / "out") + + original_cwd = Path.cwd() + monkeypatch.chdir(other_repo) + try: + result = _minimal_run_result() + orchestrator._finalize_commit_and_comparison(result, options) + finally: + monkeypatch.chdir(original_cwd) + + assert result.commit is not None + assert result.commit.sha == agent_sha + assert result.commit.sha != other_sha + + +def test_persist_resolves_commit_from_explicit_workspace( + tmp_path: Path, monkeypatch +) -> None: + for env_var in ("GITHUB_SHA", "BUILD_SOURCEVERSION", "Build.SourceVersion"): + monkeypatch.delenv(env_var, raising=False) + + agent_repo = tmp_path / "my-agent" + agent_sha = _init_repo(agent_repo, commit_subject="agent repo commit") + other_repo = tmp_path / "other-project" + _init_repo(other_repo, commit_subject="unrelated repo commit") + + output_dir = agent_repo / "out" + original_cwd = Path.cwd() + monkeypatch.chdir(other_repo) + try: + result = _minimal_run_result() + orchestrator._persist(result, output_dir, workspace=agent_repo) + finally: + monkeypatch.chdir(original_cwd) + + assert result.commit is not None + assert result.commit.sha == agent_sha + + def test_remote_dataset_is_resolved_once_and_provenance_survives_cleanup( tmp_path: Path, monkeypatch, @@ -259,3 +358,63 @@ def test_run_evaluation_preserves_dataset_semantics_across_sources( "overall_passed": True, } assert not list((tmp_path / ".agentops" / ".resolved").glob("*.jsonl")) + + +def test_persist_skips_second_commit_resolution_when_already_attempted( + tmp_path: Path, monkeypatch +) -> None: + """When ``_finalize_commit_and_comparison`` already tried (and, as here, + failed) to resolve the commit, ``_persist`` must not attempt it again - + each attempt runs real `git` subprocesses with their own timeout, so a + second identical attempt would only double the cost for the same + (failed) outcome.""" + calls = {"count": 0} + + def _fake_resolve_commit_info(*, workspace): + calls["count"] += 1 + return None + + monkeypatch.setattr( + orchestrator, "resolve_commit_info", _fake_resolve_commit_info + ) + + config_path = tmp_path / "agentops.yaml" + config_path.write_text("version: 1\n", encoding="utf-8") + options = RunOptions(config_path=config_path, output_dir=tmp_path / "out") + + result = _minimal_run_result() + orchestrator._finalize_commit_and_comparison(result, options) + assert calls["count"] == 1 + assert result.commit is None + + orchestrator._persist( + result, + options.output_dir, + workspace=options.config_path.parent, + commit_resolution_attempted=True, + ) + + assert calls["count"] == 1 + assert result.commit is None + + +def test_persist_still_attempts_commit_resolution_when_called_standalone( + tmp_path: Path, monkeypatch +) -> None: + """A caller that skips ``_finalize_commit_and_comparison`` (unlike every + current orchestrator call site) still gets the fallback, since + ``commit_resolution_attempted`` defaults to ``False``.""" + calls = {"count": 0} + + def _fake_resolve_commit_info(*, workspace): + calls["count"] += 1 + return None + + monkeypatch.setattr( + orchestrator, "resolve_commit_info", _fake_resolve_commit_info + ) + + result = _minimal_run_result() + orchestrator._persist(result, tmp_path / "out", workspace=tmp_path) + + assert calls["count"] == 1 From ec46e776260299f71a4f1d7366732dc79c865bc2 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 09:40:10 -0300 Subject: [PATCH 14/30] fix(pipeline): use the PR head SHA on pull_request events, drop invalid env var --- src/agentops/pipeline/commit_info.py | 36 +++++++++- tests/unit/test_commit_info.py | 100 ++++++++++++++++++++++++--- 2 files changed, 126 insertions(+), 10 deletions(-) diff --git a/src/agentops/pipeline/commit_info.py b/src/agentops/pipeline/commit_info.py index b0ecbc50..b0ab0f12 100644 --- a/src/agentops/pipeline/commit_info.py +++ b/src/agentops/pipeline/commit_info.py @@ -8,6 +8,7 @@ from __future__ import annotations +import json import os import subprocess from pathlib import Path @@ -15,13 +16,46 @@ from agentops.core.results import CommitInfo -_CI_SHA_ENV_VARS = ("GITHUB_SHA", "BUILD_SOURCEVERSION", "Build.SourceVersion") +_CI_SHA_ENV_VARS = ("GITHUB_SHA", "BUILD_SOURCEVERSION") _FIELD_SEP = "\x1f" _GIT_SHOW_FORMAT = f"%H{_FIELD_SEP}%h{_FIELD_SEP}%s{_FIELD_SEP}%an{_FIELD_SEP}%aI" _GIT_TIMEOUT_SECONDS = 5 +def _pull_request_head_sha() -> Optional[str]: + """The PR's head commit, read from the GitHub Actions event payload. + + On ``pull_request`` events ``GITHUB_SHA`` is the temporary merge commit + GitHub synthesizes for the PR, not the commit the PR branch is actually + at - so it must not be used to attribute a run to "the commit under + review". ``GITHUB_EVENT_PATH`` points at a JSON payload containing the + real head SHA at ``pull_request.head.sha``. + """ + event_path = os.environ.get("GITHUB_EVENT_PATH") + if not event_path: + return None + try: + with open(event_path, "r", encoding="utf-8") as handle: + event = json.load(handle) + except (OSError, ValueError): + return None + if not isinstance(event, dict): + return None + pull_request = event.get("pull_request") + if not isinstance(pull_request, dict): + return None + head = pull_request.get("head") + if not isinstance(head, dict): + return None + sha = head.get("sha") + return sha or None + + def _ci_git_sha() -> Optional[str]: + if os.environ.get("GITHUB_EVENT_NAME") == "pull_request": + head_sha = _pull_request_head_sha() + if head_sha: + return head_sha for env_var in _CI_SHA_ENV_VARS: value = os.environ.get(env_var) if value: diff --git a/tests/unit/test_commit_info.py b/tests/unit/test_commit_info.py index 45848e17..70eaee39 100644 --- a/tests/unit/test_commit_info.py +++ b/tests/unit/test_commit_info.py @@ -2,6 +2,7 @@ from __future__ import annotations +import json import subprocess from pathlib import Path @@ -70,20 +71,27 @@ def test_ci_env_var_takes_precedence_over_local_head(tmp_path, monkeypatch): def test_env_var_precedence_order(monkeypatch, tmp_path): - repo = _init_repo(tmp_path) - sha = _run(["rev-parse", "HEAD"], cwd=repo) - - monkeypatch.setenv("GITHUB_SHA", sha) - monkeypatch.setenv("BUILD_SOURCEVERSION", "some-other-sha") - monkeypatch.setenv("Build.SourceVersion", "yet-another-sha") + monkeypatch.setenv("GITHUB_SHA", "sha-from-github") + monkeypatch.setenv("BUILD_SOURCEVERSION", "sha-from-ado") - assert commit_info._ci_git_sha() == sha + assert commit_info._ci_git_sha() == "sha-from-github" monkeypatch.delenv("GITHUB_SHA", raising=False) - assert commit_info._ci_git_sha() == "some-other-sha" + assert commit_info._ci_git_sha() == "sha-from-ado" + +def test_dotted_build_sourceversion_is_not_a_real_env_var_name(monkeypatch): + """Azure Pipelines exposes ``Build.SourceVersion`` to scripts as the + env var ``BUILD_SOURCEVERSION`` (dots become underscores, uppercased) - + the literal dotted name is never actually set by the platform, so it + must not be in ``_CI_SHA_ENV_VARS`` and must not be picked up even if + something else in the environment happens to set it.""" + monkeypatch.delenv("GITHUB_SHA", raising=False) monkeypatch.delenv("BUILD_SOURCEVERSION", raising=False) - assert commit_info._ci_git_sha() == "yet-another-sha" + monkeypatch.setenv("Build.SourceVersion", "should-never-be-read") + + assert "Build.SourceVersion" not in commit_info._CI_SHA_ENV_VARS + assert commit_info._ci_git_sha() is None def test_non_git_workspace_returns_none_without_error(tmp_path, monkeypatch): @@ -135,3 +143,77 @@ def test_each_ci_env_var_is_recognized(tmp_path, monkeypatch, env_var): assert info is not None assert info.sha == sha assert info.source == "ci" + + +def _write_pull_request_event(path: Path, *, head_sha: str) -> None: + path.write_text( + json.dumps({"pull_request": {"head": {"sha": head_sha}}}), + encoding="utf-8", + ) + + +def test_pull_request_event_uses_head_sha_not_merge_commit( + tmp_path, monkeypatch +): + """On ``pull_request`` events, ``GITHUB_SHA`` is the temporary merge + commit, not the PR branch's real head - the event payload's + ``pull_request.head.sha`` must win instead.""" + repo = _init_repo(tmp_path) + head_sha = _run(["rev-parse", "HEAD"], cwd=repo) + merge_commit_sha = "f" * 40 + + event_path = tmp_path / "event.json" + _write_pull_request_event(event_path, head_sha=head_sha) + + monkeypatch.setenv("GITHUB_EVENT_NAME", "pull_request") + monkeypatch.setenv("GITHUB_EVENT_PATH", str(event_path)) + monkeypatch.setenv("GITHUB_SHA", merge_commit_sha) + monkeypatch.delenv("BUILD_SOURCEVERSION", raising=False) + monkeypatch.delenv("Build.SourceVersion", raising=False) + + info = commit_info.resolve_commit_info(workspace=repo) + + assert info is not None + assert info.sha == head_sha + assert info.sha != merge_commit_sha + assert info.source == "ci" + + +def test_pull_request_event_falls_back_to_github_sha_when_payload_unusable( + tmp_path, monkeypatch +): + repo = _init_repo(tmp_path) + sha = _run(["rev-parse", "HEAD"], cwd=repo) + + monkeypatch.setenv("GITHUB_EVENT_NAME", "pull_request") + monkeypatch.setenv("GITHUB_EVENT_PATH", str(tmp_path / "missing.json")) + monkeypatch.setenv("GITHUB_SHA", sha) + monkeypatch.delenv("BUILD_SOURCEVERSION", raising=False) + monkeypatch.delenv("Build.SourceVersion", raising=False) + + info = commit_info.resolve_commit_info(workspace=repo) + + assert info is not None + assert info.sha == sha + assert info.source == "ci" + + +def test_non_pull_request_event_still_uses_github_sha_directly( + tmp_path, monkeypatch +): + """Push events have no ``pull_request`` payload to read - GITHUB_SHA is + already the right commit and must be used as-is.""" + repo = _init_repo(tmp_path) + sha = _run(["rev-parse", "HEAD"], cwd=repo) + + monkeypatch.setenv("GITHUB_EVENT_NAME", "push") + monkeypatch.delenv("GITHUB_EVENT_PATH", raising=False) + monkeypatch.setenv("GITHUB_SHA", sha) + monkeypatch.delenv("BUILD_SOURCEVERSION", raising=False) + monkeypatch.delenv("Build.SourceVersion", raising=False) + + info = commit_info.resolve_commit_info(workspace=repo) + + assert info is not None + assert info.sha == sha + assert info.source == "ci" From 71ff7ae2ea17a3c45320ab203a3f92d14da178a7 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 09:40:35 -0300 Subject: [PATCH 15/30] feat(pipeline): list every regressed metric in the insight, not just the worst --- docs/how-it-works.md | 40 ++-- .../contracts/results-json.md | 26 ++- .../data-model.md | 12 +- .../quickstart.md | 4 +- src/agentops/core/results.py | 36 +++- src/agentops/pipeline/comparison.py | 53 +++-- src/agentops/pipeline/regression_insight.py | 199 ++++++++++++++---- src/agentops/pipeline/reporter.py | 11 + .../test_regression_commit_attribution.py | 9 +- tests/unit/test_pipeline_comparison.py | 74 ++++++- tests/unit/test_pipeline_reporter.py | 73 ++++++- tests/unit/test_regression_insight.py | 188 +++++++++++++++-- 12 files changed, 599 insertions(+), 126 deletions(-) diff --git a/docs/how-it-works.md b/docs/how-it-works.md index 3cb45d32..c56a56d2 100644 --- a/docs/how-it-works.md +++ b/docs/how-it-works.md @@ -725,21 +725,35 @@ Foundry prompt-agent deploy gating), and on a best-effort basis locally via `git rev-parse HEAD`. It's `null` when the workspace isn't a git repository or `git` is unavailable; nothing else about the run changes. -When a metric regresses between two comparable runs (same agent target, -dataset, and evaluator set) that both have commit metadata - whether -detected via an explicit `--baseline` comparison or Doctor's rolling -regression check - `report.md` gains a "Regression Insight" section -explaining what changed between the two commits (system prompt, model, -dataset, evaluators, or thresholds) and suggesting a corrective action. This -is deterministic field-diffing, not an LLM call, and it's purely -informational: it never affects the exit-code/threshold-gating contract. -Runs without commit metadata, or where nothing tracked changed, behave -exactly as they did before this existed. +When one or more metrics regress between two comparable runs (same agent +target, dataset, and evaluator set) that both have commit metadata - +whether detected via an explicit `--baseline` comparison or Doctor's +rolling regression check - `report.md` gains a "Regression Insight" section +listing every metric that regressed (not just the worst one) and explaining +what changed between the two commits (system prompt, model, dataset, +evaluators, or thresholds) and suggesting a corrective action. When either +run was published to Foundry (`execution: cloud`, or local `publish: true`), +the section also links out to its Evaluations page. This is deterministic +field-diffing, not an LLM call, and it's purely informational: it never +affects the exit-code/threshold-gating contract. Runs without commit +metadata, or where nothing tracked changed, behave exactly as they did +before this existed. + +Every run produced by `agentops eval run` - including `execution: cloud` +and `execution: azd` - attempts commit capture. The one case with no commit +to attribute is Doctor's rolling regression check when it falls back to +Foundry cloud evaluation runs because local history is too short (see +`results_history`'s cloud fallback): those runs have no local +`results.json` to reload and therefore no recorded commit. When that +happens, the regression finding's recommendation says so explicitly (e.g. +`attribution unavailable: run has no commit (fetched from cloud)`) instead +of silently omitting the insight. Cockpit's dashboard also gains an "Evaluation Version History" section -listing every locally recorded run, newest first, with its commit and what -changed relative to the previous run in its lineage - shown regardless of -whether that run regressed, so you can browse the causal trail over time. +listing every locally recorded run, newest first, with its commit, which +metrics (if any) regressed relative to the previous run in its lineage by +name, and what changed - shown regardless of whether that run regressed, so +you can browse the causal trail over time. ## Testing diff --git a/specs/012-regression-commit-attribution/contracts/results-json.md b/specs/012-regression-commit-attribution/contracts/results-json.md index eb756624..0b459d67 100644 --- a/specs/012-regression-commit-attribution/contracts/results-json.md +++ b/specs/012-regression-commit-attribution/contracts/results-json.md @@ -29,8 +29,9 @@ MUST treat a missing/`null` `commit` the same as before this feature existed. ## New field on `comparison`: `insight` Only present when `comparison` (the existing `--baseline` block) is present -**and** a regression was detected **and** both the current and baseline runs -have non-null `commit`: +**and** at least one metric regressed **and** both the current and baseline +runs have non-null `commit`. `regressed_metrics` lists every metric that +regressed, not just the worst one, ordered worst-first: ```jsonc { @@ -41,25 +42,30 @@ have non-null `commit`: "to_run_id": "20260910-140300", "from_commit": { "sha": "...", "short_sha": "...", "subject": "...", "author": "...", "authored_at": "...", "source": "ci" }, "to_commit": { "sha": "...", "short_sha": "...", "subject": "...", "author": "...", "authored_at": "...", "source": "ci" }, - "metric": "accuracy", - "from_value": 0.91, - "to_value": 0.79, + "regressed_metrics": [ + { "metric": "accuracy", "from_value": 0.91, "to_value": 0.79 } + ], "changed_inputs": [ { "field": "system_prompt", "description": "the system prompt changed", "from_value": "greeter:v3", "to_value": "greeter:v4" }, { "field": "model", "description": "model changed from gpt-4o to gpt-4o-mini", "from_value": "gpt-4o", "to_value": "gpt-4o-mini" } ], "explanation": "Run v3 → v4: accuracy dropped from 0.91 to 0.79. Likely cause: the system prompt changed and the model changed from gpt-4o to gpt-4o-mini.", "suggested_action": "Review the prompt change and the model swap; consider reverting one at a time to isolate the cause.", - "used_git_diff": false + "commits_available_locally": false, + "from_report_url": null, + "to_report_url": null } } } ``` -Absent (`comparison.insight` key not present) whenever no regression was -detected, either compared run lacks commit metadata, or `comparison` itself -is absent (no `--baseline` was used). This preserves every existing -`comparison`-shaped consumer. +Serialized as `comparison.insight: null` (the key is present, since +`model_dump(mode="json")` is called without `exclude_none`) whenever no +metric regressed or either compared run lacks commit metadata. When +`comparison` itself is absent (no `--baseline` was used), there is no +nested `insight` field at all. Either way, every existing +`comparison`-shaped consumer that ignores unknown-to-it or null fields is +unaffected. ## No changes to exit codes or CLI flags diff --git a/specs/012-regression-commit-attribution/data-model.md b/specs/012-regression-commit-attribution/data-model.md index 3c02195d..9aecb663 100644 --- a/specs/012-regression-commit-attribution/data-model.md +++ b/specs/012-regression-commit-attribution/data-model.md @@ -43,13 +43,15 @@ The causal explanation for a detected regression between two comparable runs. | `to_run_id` | `str` | Identifier of the regressed run. | | `from_commit` | `Optional[CommitInfo]` | Prior run's commit, if known. | | `to_commit` | `Optional[CommitInfo]` | Regressed run's commit, if known. | -| `metric` | `str` | The regressed metric's name. | -| `from_value` | `float` | Metric value on the prior run. | -| `to_value` | `float` | Metric value on the regressed run. | +| `regressed_metrics` | `List[RegressedMetric]` | Every metric that regressed between the two runs, not just the worst one, ordered worst-first (direction-aware). Each entry: `{metric: str, from_value: float, to_value: float}`. | | `changed_inputs` | `List[ChangedInput]` | All detected changes, not just the first found (per spec edge case). | -| `explanation` | `str` | The rendered one-to-two-sentence plain-language summary (FR-007). | +| `explanation` | `str` | The rendered plain-language summary covering every entry in `regressed_metrics` (FR-007). | | `suggested_action` | `Optional[str]` | Brief, rule-based corrective suggestion (FR-008). | -| `used_git_diff` | `bool` | `True` when both commits were reachable in local git history for a fuller diff; `False` when this fell back to comparing only the fields already recorded in each run's stored result (research.md #4). | +| `commits_available_locally` | `bool` | `True` when both commits were reachable in local git history, `False` otherwise (e.g. a shallow CI checkout per research.md #4). Informational only - the comparison itself is always the field-based diff described in research.md #4/#5, never a tree-level `git diff`, regardless of this value. | +| `from_report_url` | `Optional[str]` | Foundry Evaluations deep-link for the prior run, when it was published (`execution: cloud`, or local `publish: true` after its publish step completed). `None` when not published - never fabricated. | +| `to_report_url` | `Optional[str]` | Same, for the regressed run. | + +A `RegressedMetric` has `metric: str`, `from_value: float`, `to_value: float`. **Validation rule**: A `RegressionInsight` is only ever constructed when both `from_commit` and `to_commit` are non-`None` (FR-006, FR-009). If either run diff --git a/specs/012-regression-commit-attribution/quickstart.md b/specs/012-regression-commit-attribution/quickstart.md index 7fe5b872..44d5c471 100644 --- a/specs/012-regression-commit-attribution/quickstart.md +++ b/specs/012-regression-commit-attribution/quickstart.md @@ -82,8 +82,8 @@ fallback, missing-git/non-repo handling. - `tests/unit/test_regression_insight.py` — field-diff detection (prompt, model, dataset, evaluators, thresholds), the "both commits missing" and - "one commit missing" no-fabrication paths, and the `used_git_diff` - fallback branch. + "one commit missing" no-fabrication paths, and the + `commits_available_locally` fallback branch. - `tests/unit/test_reporter.py` — new "Regression Insight" section rendering, and its absence when `insight` is `None`. - `tests/unit/test_cockpit.py` — `_project_run` commit/`changed_inputs` diff --git a/src/agentops/core/results.py b/src/agentops/core/results.py index 7d348c61..1441f424 100644 --- a/src/agentops/core/results.py +++ b/src/agentops/core/results.py @@ -107,20 +107,46 @@ class ChangedInput(BaseModel): to_value: Optional[str] = None +class RegressedMetric(BaseModel): + """One metric's before/after values in a detected regression.""" + + metric: str + from_value: float + to_value: float + + class RegressionInsight(BaseModel): - """Causal explanation for a detected regression between two runs.""" + """Causal explanation for a detected regression between two runs. + + Lists every metric that regressed between the two runs, not just the + single worst one - a prompt/model/dataset change between two runs + commonly moves several metrics at once, and the other surfaces (Doctor + reports one finding per metric already) shouldn't be narrower than that. + """ from_run_id: str to_run_id: str from_commit: Optional[CommitInfo] = None to_commit: Optional[CommitInfo] = None - metric: str - from_value: float - to_value: float + # Ordered worst-first (direction-aware - see + # `pipeline.regression_insight.regression_severity`). + regressed_metrics: List[RegressedMetric] changed_inputs: List[ChangedInput] = Field(default_factory=list) explanation: str suggested_action: Optional[str] = None - used_git_diff: bool = False + # Whether both commits were present in local git history. Does not mean + # a `git diff` of either tree was run - the explanation above is always + # built purely from fields already recorded on each run's RunResult (see + # `pipeline.regression_insight`'s module docstring). True only means a + # fuller history-based diff would have been *possible*. + commits_available_locally: bool = False + # Deep-link to the Foundry Evaluations page for each run, when it was + # published there (`execution: cloud`, or local `publish: true` after + # its Classic Foundry publish step has completed) - see + # `pipeline.regression_insight.resolve_report_url`. Informational only; + # `None` whenever a run wasn't published, never fabricated. + from_report_url: Optional[str] = None + to_report_url: Optional[str] = None class ComparisonInfo(BaseModel): diff --git a/src/agentops/pipeline/comparison.py b/src/agentops/pipeline/comparison.py index efb99afb..11da57e8 100644 --- a/src/agentops/pipeline/comparison.py +++ b/src/agentops/pipeline/comparison.py @@ -12,7 +12,17 @@ ComparisonRow, RunResult, ) -from agentops.pipeline.regression_insight import build_regression_insight +from agentops.pipeline.regression_insight import ( + LOWER_IS_BETTER_METRICS, + build_regression_insight, + metric_improved, + resolve_report_url, +) + +# Re-exported for callers that only need the metric-direction concept +# without pulling in regression-insight building - the single source of +# truth lives in `regression_insight` (see its module comment for why). +__all__ = ["LOWER_IS_BETTER_METRICS", "metric_improved", "load_baseline", "build_comparison"] def load_baseline(path: Path) -> RunResult: @@ -24,14 +34,11 @@ def load_baseline(path: Path) -> RunResult: return RunResult.model_validate(payload) -def _direction(current: Optional[float], baseline: Optional[float]) -> str: - if current is None or baseline is None: +def _direction(metric: str, current: Optional[float], baseline: Optional[float]) -> str: + improved = metric_improved(metric, current, baseline) + if improved is None: return "unchanged" - if current > baseline: - return "improved" - if current < baseline: - return "regressed" - return "unchanged" + return "improved" if improved else "regressed" def _row_passed(row_metrics: List[Dict[str, float | None]]) -> bool: @@ -39,23 +46,12 @@ def _row_passed(row_metrics: List[Dict[str, float | None]]) -> bool: return all("error" not in metric or not metric["error"] for metric in row_metrics) -def _relative_drop(metric: ComparisonMetric) -> float: - """Fraction the metric dropped relative to baseline (same formula as - ``agent.checks.regression``'s rolling-baseline drop calculation), used to - pick which regressed metric to attribute a cause to when several - regressed at once. Non-positive or missing baselines can't produce a - meaningful ratio, so they sort last rather than raising. - """ - if metric.baseline is None or metric.current is None or metric.baseline <= 0: - return 0.0 - return (metric.baseline - metric.current) / metric.baseline - - def build_comparison( *, current: RunResult, baseline: RunResult, baseline_path: Path, + workspace: Optional[Path] = None, ) -> ComparisonInfo: metrics: List[ComparisonMetric] = [] metric_names = sorted(set(current.aggregate_metrics) | set(baseline.aggregate_metrics)) @@ -73,7 +69,7 @@ def build_comparison( current=current_value, baseline=baseline_value, delta=delta, - direction=_direction(current_value, baseline_value), + direction=_direction(name, current_value, baseline_value), ) ) @@ -116,8 +112,19 @@ def build_comparison( if current.commit is not None and baseline.commit is not None: regressed = [m for m in metrics if m.direction == "regressed"] if regressed: - worst = max(regressed, key=_relative_drop) - insight = build_regression_insight(baseline, current, metric=worst.metric) + insight = build_regression_insight( + baseline, + current, + metrics=[m.metric for m in regressed], + workspace=workspace, + # `baseline` is a prior, fully-published run - a sidecar + # `cloud_evaluation.json` next to it, if any, is complete. + # `current` is still mid-orchestration here (persisted + # after this call, Classic Foundry `publish: true` later + # still) - only its own in-memory config can be checked. + from_report_url=resolve_report_url(baseline, results_path=baseline_path), + to_report_url=resolve_report_url(current), + ) return ComparisonInfo( baseline_path=str(baseline_path), diff --git a/src/agentops/pipeline/regression_insight.py b/src/agentops/pipeline/regression_insight.py index fd625242..20d9b484 100644 --- a/src/agentops/pipeline/regression_insight.py +++ b/src/agentops/pipeline/regression_insight.py @@ -8,13 +8,26 @@ ``RunResult`` (target, config, dataset, evaluators, thresholds) - no git tree access, no network calls, no LLM calls, per the feature's determinism requirement. + +Caveat: a Foundry prompt-agent system-prompt change is only detected via a +``name:version`` bump (see ``_target_changes``) - editing a prompt's content +without publishing a new version produces no ``system_prompt`` entry in +``changed_inputs``, since no prompt text or content hash is captured on +``RunResult`` to diff against. """ from __future__ import annotations +import json +from pathlib import Path from typing import Any, Dict, List, Optional -from agentops.core.results import ChangedInput, RegressionInsight, RunResult +from agentops.core.results import ( + ChangedInput, + RegressedMetric, + RegressionInsight, + RunResult, +) from agentops.pipeline.commit_info import commit_exists_locally _TRACKED_CONFIG_FIELDS = ("dataset", "evaluators", "thresholds") @@ -27,6 +40,85 @@ "thresholds": "threshold change", } +# Metrics where a lower value is the better outcome. Every other metric +# (and every evaluator's default threshold - see `core/evaluators.py`, +# where this is the only metric with a `<=` default) is treated as +# higher-is-better. Lives here (rather than `pipeline.comparison`, its only +# other consumer) because this module has no dependency on `comparison`, +# while `comparison` already depends on this module for +# `build_regression_insight` - putting it here avoids a circular import. +LOWER_IS_BETTER_METRICS = frozenset({"avg_latency_seconds"}) + + +def metric_improved( + metric: str, current: Optional[float], baseline: Optional[float] +) -> Optional[bool]: + """Whether ``current`` is a better outcome than ``baseline`` for ``metric``. + + Returns ``None`` when the values are missing or equal (no judgement to + make), ``True`` when ``current`` is the better value, ``False`` when + it's worse - accounting for ``metric`` being lower-is-better (e.g. + latency) vs. the higher-is-better default. + """ + if current is None or baseline is None or current == baseline: + return None + if metric in LOWER_IS_BETTER_METRICS: + return current < baseline + return current > baseline + + +def regression_severity(metric: str, from_value: float, to_value: float) -> float: + """How severely ``metric`` regressed from ``from_value`` to ``to_value``, + as a fraction of ``from_value`` when that's meaningful, else as an + absolute change. Used only to order ``RegressionInsight.regressed_metrics`` + worst-first - direction-aware, so a lower-is-better metric (e.g. + latency) getting worse produces a positive value just like any other + regression. A ``from_value`` at or below zero can't produce a + meaningful ratio, so the absolute change is used instead of collapsing + to zero. + """ + worse_by = ( + to_value - from_value + if metric in LOWER_IS_BETTER_METRICS + else from_value - to_value + ) + if worse_by <= 0: + return 0.0 + if from_value <= 0: + return worse_by + return worse_by / from_value + + +def resolve_report_url(run: RunResult, *, results_path: Optional[Path] = None) -> Optional[str]: + """The Foundry Evaluations deep-link for ``run``, when it was published. + + Checks the run's own recorded config first (``execution: cloud`` embeds + ``report_url`` there before ``results.json`` is even written), then - + when ``results_path`` is given - a sibling ``cloud_evaluation.json`` + (written by a completed local ``publish: true`` Classic Foundry publish + step, which happens *after* ``results.json`` and so is never in the + run's own config). Returns ``None`` - never raises - whenever neither + source has a URL, e.g. the run was never published. + """ + cloud_evaluation = (run.config or {}).get("cloud_evaluation") + if isinstance(cloud_evaluation, dict): + url = cloud_evaluation.get("report_url") + if isinstance(url, str) and url: + return url + + if results_path is not None: + sidecar = results_path.parent / "cloud_evaluation.json" + try: + data = json.loads(sidecar.read_text(encoding="utf-8")) + except (OSError, ValueError): + return None + if isinstance(data, dict): + url = data.get("report_url") + if isinstance(url, str) and url: + return url + + return None + def _target_changes(from_run: RunResult, to_run: RunResult) -> List[ChangedInput]: changes: List[ChangedInput] = [] @@ -125,30 +217,36 @@ def build_changed_inputs(from_run: RunResult, to_run: RunResult) -> List[Changed return _target_changes(from_run, to_run) + _config_changes(from_run, to_run) +def _join_with_and(items: List[str]) -> str: + if len(items) == 1: + return items[0] + if len(items) == 2: + return f"{items[0]} and {items[1]}" + return ", ".join(items[:-1]) + f", and {items[-1]}" + + +def _describe_metric_change(metric: RegressedMetric) -> str: + # Every entry here is, by construction, a regression - and for both a + # higher-is-better metric dropping and a lower-is-better metric (e.g. + # latency) rising, "got worse" always matches this raw numeric + # comparison, so no direction lookup is needed just to word it. + verb = "dropped" if metric.to_value < metric.from_value else "increased" + return f"{metric.metric} {verb} from {metric.from_value:.2f} to {metric.to_value:.2f}" + + def _explanation( *, - metric: str, - from_value: float, - to_value: float, + regressed_metrics: List[RegressedMetric], changed_inputs: List[ChangedInput], from_short_sha: str, to_short_sha: str, ) -> str: - direction = "dropped" if to_value < from_value else "changed" - base = ( - f"Run {from_short_sha} → {to_short_sha}: {metric} {direction} " - f"from {from_value:.2f} to {to_value:.2f}." - ) + metrics_text = _join_with_and([_describe_metric_change(m) for m in regressed_metrics]) + base = f"Run {from_short_sha} → {to_short_sha}: {metrics_text}." if not changed_inputs: return base - descriptions = [c.description for c in changed_inputs] - if len(descriptions) == 1: - cause = descriptions[0] - elif len(descriptions) == 2: - cause = f"{descriptions[0]} and {descriptions[1]}" - else: - cause = ", ".join(descriptions[:-1]) + f", and {descriptions[-1]}" + cause = _join_with_and([c.description for c in changed_inputs]) return f"{base} Likely cause: {cause}." @@ -158,54 +256,81 @@ def _suggested_action(changed_inputs: List[ChangedInput]) -> Optional[str]: labels = [_SUGGESTION_LABELS.get(c.field, f"{c.field} change") for c in changed_inputs] if len(labels) == 1: return f"Review the {labels[0]} to confirm it's the cause, and revert it if so." - joined = ", ".join(labels[:-1]) + f", and {labels[-1]}" if len(labels) > 2 else " and ".join(labels) - return f"Review the {joined}; consider reverting one at a time to isolate the cause." + return f"Review the {_join_with_and(labels)}; consider reverting one at a time to isolate the cause." def build_regression_insight( from_run: RunResult, to_run: RunResult, *, - metric: str, + metrics: List[str], + workspace: Optional[Path] = None, + from_report_url: Optional[str] = None, + to_report_url: Optional[str] = None, ) -> Optional[RegressionInsight]: - """Explain a regression on ``metric`` between two comparable runs. + """Explain a regression on ``metrics`` between two comparable runs. + + ``metrics`` is every metric known to have regressed between the two + runs (not just the worst one) - all of them are listed in the returned + insight's ``regressed_metrics``, ordered worst-first + (``regression_severity``), so the explanation and the report/Cockpit + surfaces consuming it don't silently drop all but one when several + metrics move together. Returns ``None`` (produces no fabricated cause) when either run lacks - commit metadata or the metric's value isn't present on both runs, - per FR-009. Otherwise diffs the runs' recorded fields (see - ``build_changed_inputs``) and renders a plain-language explanation plus - a short suggested corrective action. + commit metadata, or when none of ``metrics`` has a value on both runs, + per FR-009 - an individual metric missing a value on either side is + just skipped rather than failing the whole insight. Otherwise diffs the + runs' recorded fields (see ``build_changed_inputs``) and renders a + plain-language explanation plus a short suggested corrective action. + + ``workspace`` is the directory the two commits' repository lives in; + it's forwarded to the local git lookups below so they run against the + right checkout instead of the current process's cwd. ``from_report_url`` + /``to_report_url`` (see ``resolve_report_url``) are carried through + as-is - this function does no file I/O to resolve them itself, since + only the caller knows whether a run's ``results.json`` path is + available to check for a sidecar ``cloud_evaluation.json``. """ if from_run.commit is None or to_run.commit is None: return None - from_value = from_run.aggregate_metrics.get(metric) - to_value = to_run.aggregate_metrics.get(metric) - if from_value is None or to_value is None: + regressed_metrics: List[RegressedMetric] = [] + for metric in metrics: + from_value = from_run.aggregate_metrics.get(metric) + to_value = to_run.aggregate_metrics.get(metric) + if from_value is None or to_value is None: + continue + regressed_metrics.append( + RegressedMetric(metric=metric, from_value=from_value, to_value=to_value) + ) + if not regressed_metrics: return None + regressed_metrics.sort( + key=lambda m: regression_severity(m.metric, m.from_value, m.to_value), + reverse=True, + ) changed_inputs = build_changed_inputs(from_run, to_run) - used_git_diff = commit_exists_locally(from_run.commit.sha) and commit_exists_locally( - to_run.commit.sha - ) + commits_available_locally = commit_exists_locally( + from_run.commit.sha, workspace=workspace + ) and commit_exists_locally(to_run.commit.sha, workspace=workspace) return RegressionInsight( from_run_id=from_run.started_at, to_run_id=to_run.started_at, from_commit=from_run.commit, to_commit=to_run.commit, - metric=metric, - from_value=from_value, - to_value=to_value, + regressed_metrics=regressed_metrics, changed_inputs=changed_inputs, explanation=_explanation( - metric=metric, - from_value=from_value, - to_value=to_value, + regressed_metrics=regressed_metrics, changed_inputs=changed_inputs, from_short_sha=from_run.commit.short_sha, to_short_sha=to_run.commit.short_sha, ), suggested_action=_suggested_action(changed_inputs), - used_git_diff=used_git_diff, + commits_available_locally=commits_available_locally, + from_report_url=from_report_url, + to_report_url=to_report_url, ) diff --git a/src/agentops/pipeline/reporter.py b/src/agentops/pipeline/reporter.py index af2f020b..bebcbf69 100644 --- a/src/agentops/pipeline/reporter.py +++ b/src/agentops/pipeline/reporter.py @@ -228,10 +228,21 @@ def _render_comparison(comparison: ComparisonInfo) -> List[str]: def _render_regression_insight(insight: RegressionInsight) -> List[str]: lines = ["## Regression Insight", ""] + if len(insight.regressed_metrics) > 1: + lines.append("**Regressed metrics:**") + for rm in insight.regressed_metrics: + lines.append(f"- `{rm.metric}`: {rm.from_value:.3f} → {rm.to_value:.3f}") + lines.append("") lines.append(_short(insight.explanation, 500)) if insight.suggested_action: lines.append("") lines.append(f"**Suggested action:** {_short(insight.suggested_action, 300)}") + if insight.from_report_url or insight.to_report_url: + lines.append("") + if insight.from_report_url: + lines.append(f"- [Baseline run in Foundry]({insight.from_report_url})") + if insight.to_report_url: + lines.append(f"- [Regressed run in Foundry]({insight.to_report_url})") return lines diff --git a/tests/integration/test_regression_commit_attribution.py b/tests/integration/test_regression_commit_attribution.py index 6de7c730..3a4c0e12 100644 --- a/tests/integration/test_regression_commit_attribution.py +++ b/tests/integration/test_regression_commit_attribution.py @@ -128,7 +128,9 @@ def _run_baseline_then_regressed(tmp_path: Path, monkeypatch, good_url: str, bad _write_dataset(current_dataset) commits = iter([_fake_commit("a" * 40), _fake_commit("b" * 40)]) - monkeypatch.setattr(orchestrator, "resolve_commit_info", lambda: next(commits)) + monkeypatch.setattr( + orchestrator, "resolve_commit_info", lambda **kwargs: next(commits) + ) baseline_config_path = tmp_path / "agentops-baseline.yaml" _write_config(baseline_config_path, agent_url=good_url, dataset=baseline_dataset) @@ -180,8 +182,9 @@ def test_regression_commit_attribution_end_to_end( assert current_result.comparison.insight is not None insight = current_result.comparison.insight - assert insight.metric == "f1_score" - assert insight.from_value == pytest.approx(1.0) + assert len(insight.regressed_metrics) == 1 + assert insight.regressed_metrics[0].metric == "f1_score" + assert insight.regressed_metrics[0].from_value == pytest.approx(1.0) assert any(c.field == "dataset" for c in insight.changed_inputs) report_text = (current_dir / "report.md").read_text(encoding="utf-8") diff --git a/tests/unit/test_pipeline_comparison.py b/tests/unit/test_pipeline_comparison.py index b9a8de14..1a6032e3 100644 --- a/tests/unit/test_pipeline_comparison.py +++ b/tests/unit/test_pipeline_comparison.py @@ -25,11 +25,14 @@ def _run( deployment: str | None = "gpt-4o", accuracy: float, coherence: float | None = None, + avg_latency_seconds: float | None = None, commit: CommitInfo | None, ) -> RunResult: metrics = {"accuracy": accuracy} if coherence is not None: metrics["coherence"] = coherence + if avg_latency_seconds is not None: + metrics["avg_latency_seconds"] = avg_latency_seconds return RunResult( started_at="2026-09-01T10:00:00+00:00", finished_at="2026-09-01T10:00:01+00:00", @@ -66,9 +69,10 @@ def test_build_comparison_attaches_insight_when_regressed_and_commits_known(): ) assert info.insight is not None - assert info.insight.metric == "accuracy" - assert info.insight.from_value == 0.91 - assert info.insight.to_value == 0.79 + assert len(info.insight.regressed_metrics) == 1 + assert info.insight.regressed_metrics[0].metric == "accuracy" + assert info.insight.regressed_metrics[0].from_value == 0.91 + assert info.insight.regressed_metrics[0].to_value == 0.79 def test_build_comparison_no_insight_without_commit_metadata(): @@ -82,10 +86,11 @@ def test_build_comparison_no_insight_without_commit_metadata(): assert info.insight is None -def test_build_comparison_picks_worst_relative_drop_not_alphabetical_first(): - """`accuracy` sorts before `coherence` alphabetically, but `coherence` - dropped much more in relative terms (56% vs 13%) - the insight must be - for `coherence`, not whichever metric name comes first. +def test_build_comparison_lists_every_regressed_metric_ordered_worst_first(): + """`accuracy` and `coherence` both regress at once (13% and 56% + respectively) - both must appear in the insight, worst first, not just + whichever metric name sorts first alphabetically or a single "worst" + one with the rest silently dropped. """ baseline = _run( version="3", deployment="gpt-4o", accuracy=0.91, coherence=4.5, commit=_commit("a" * 40) @@ -99,7 +104,7 @@ def test_build_comparison_picks_worst_relative_drop_not_alphabetical_first(): ) assert info.insight is not None - assert info.insight.metric == "coherence" + assert [m.metric for m in info.insight.regressed_metrics] == ["coherence", "accuracy"] def test_build_comparison_no_insight_when_nothing_regressed(): @@ -111,3 +116,56 @@ def test_build_comparison_no_insight_when_nothing_regressed(): ) assert info.insight is None + + +def test_latency_drop_is_improved_not_regressed(): + """``avg_latency_seconds`` is lower-is-better - a drop from 8s to 3s is + an improvement, not a regression, even though the raw value went down.""" + baseline = _run(accuracy=0.91, avg_latency_seconds=8.0, commit=_commit("a" * 40)) + current = _run(accuracy=0.91, avg_latency_seconds=3.0, commit=_commit("b" * 40)) + + info = comparison.build_comparison( + current=current, baseline=baseline, baseline_path=Path(".agentops/baseline/results.json") + ) + + latency_metric = next(m for m in info.metrics if m.metric == "avg_latency_seconds") + assert latency_metric.direction == "improved" + assert info.insight is None + + +def test_latency_increase_is_regressed_and_explained(): + baseline = _run(accuracy=0.91, avg_latency_seconds=3.0, commit=_commit("a" * 40)) + current = _run(accuracy=0.91, avg_latency_seconds=8.0, commit=_commit("b" * 40)) + + info = comparison.build_comparison( + current=current, baseline=baseline, baseline_path=Path(".agentops/baseline/results.json") + ) + + latency_metric = next(m for m in info.metrics if m.metric == "avg_latency_seconds") + assert latency_metric.direction == "regressed" + assert info.insight is not None + assert info.insight.regressed_metrics[0].metric == "avg_latency_seconds" + + +def test_insight_carries_baseline_report_url_from_sidecar_file(tmp_path: Path): + """``baseline`` is a prior, fully-published run - a sidecar + ``cloud_evaluation.json`` next to its ``results.json`` (as a completed + local ``publish: true`` run would have) must surface as + ``from_report_url``. ``current`` is still mid-orchestration (no file on + disk yet), so ``to_report_url`` stays ``None`` unless its own in-memory + config already has it (``execution: cloud``).""" + baseline_dir = tmp_path / "baseline" + baseline_dir.mkdir() + baseline_path = baseline_dir / "results.json" + (baseline_dir / "cloud_evaluation.json").write_text( + '{"report_url": "https://ai.azure.com/foundry/baseline"}', encoding="utf-8" + ) + + baseline = _run(version="3", deployment="gpt-4o", accuracy=0.91, commit=_commit("a" * 40)) + current = _run(version="4", deployment="gpt-4o-mini", accuracy=0.79, commit=_commit("b" * 40)) + + info = comparison.build_comparison(current=current, baseline=baseline, baseline_path=baseline_path) + + assert info.insight is not None + assert info.insight.from_report_url == "https://ai.azure.com/foundry/baseline" + assert info.insight.to_report_url is None diff --git a/tests/unit/test_pipeline_reporter.py b/tests/unit/test_pipeline_reporter.py index 1883140f..67fb1924 100644 --- a/tests/unit/test_pipeline_reporter.py +++ b/tests/unit/test_pipeline_reporter.py @@ -7,6 +7,7 @@ CommitInfo, ComparisonInfo, ComparisonMetric, + RegressedMetric, RegressionInsight, RowMetric, RowResult, @@ -137,9 +138,9 @@ def test_report_renders_regression_insight_section_when_present(): to_run_id="2026-09-10T14:03:00+00:00", from_commit=_commit("a" * 40), to_commit=_commit("b" * 40), - metric="accuracy", - from_value=0.91, - to_value=0.79, + regressed_metrics=[ + RegressedMetric(metric="accuracy", from_value=0.91, to_value=0.79) + ], changed_inputs=[ ChangedInput(field="model", description="the model changed from gpt-4o to gpt-4o-mini") ], @@ -148,7 +149,7 @@ def test_report_renders_regression_insight_section_when_present(): "Likely cause: the model changed from gpt-4o to gpt-4o-mini." ), suggested_action="Review the model change; consider reverting it.", - used_git_diff=False, + commits_available_locally=False, ), ) @@ -162,6 +163,70 @@ def test_report_renders_regression_insight_section_when_present(): assert text.index("## Comparison vs Baseline") < text.index("## Regression Insight") +def test_report_lists_every_regressed_metric_and_report_urls(): + result = _result() + result.comparison = ComparisonInfo( + baseline_path=".agentops/baseline/results.json", + metrics=[ + ComparisonMetric( + metric="coherence", current=2.0, baseline=4.5, delta=-2.5, direction="regressed" + ), + ComparisonMetric( + metric="similarity", current=3.0, baseline=4.0, delta=-1.0, direction="regressed" + ), + ], + insight=RegressionInsight( + from_run_id="2026-09-01T10:00:00+00:00", + to_run_id="2026-09-10T14:03:00+00:00", + from_commit=_commit("a" * 40), + to_commit=_commit("b" * 40), + regressed_metrics=[ + RegressedMetric(metric="coherence", from_value=4.5, to_value=2.0), + RegressedMetric(metric="similarity", from_value=4.0, to_value=3.0), + ], + changed_inputs=[], + explanation="Run aaaaaaa → bbbbbbb: coherence dropped from 4.50 to 2.00 and similarity dropped from 4.00 to 3.00.", + commits_available_locally=False, + from_report_url="https://ai.azure.com/foundry/baseline", + to_report_url="https://ai.azure.com/foundry/current", + ), + ) + + text = reporter.render(result) + + assert "**Regressed metrics:**" in text + assert "`coherence`: 4.500 → 2.000" in text + assert "`similarity`: 4.000 → 3.000" in text + assert "[Baseline run in Foundry](https://ai.azure.com/foundry/baseline)" in text + assert "[Regressed run in Foundry](https://ai.azure.com/foundry/current)" in text + + +def test_report_omits_report_url_links_when_absent(): + result = _result() + result.comparison = ComparisonInfo( + baseline_path=".agentops/baseline/results.json", + metrics=[ + ComparisonMetric( + metric="accuracy", current=0.79, baseline=0.91, delta=-0.12, direction="regressed" + ) + ], + insight=RegressionInsight( + from_run_id="2026-09-01T10:00:00+00:00", + to_run_id="2026-09-10T14:03:00+00:00", + from_commit=_commit("a" * 40), + to_commit=_commit("b" * 40), + regressed_metrics=[RegressedMetric(metric="accuracy", from_value=0.91, to_value=0.79)], + changed_inputs=[], + explanation="Run aaaaaaa → bbbbbbb: accuracy dropped from 0.91 to 0.79.", + commits_available_locally=False, + ), + ) + + text = reporter.render(result) + + assert "Foundry]" not in text + + def test_report_has_no_regression_insight_section_when_absent(): result = _result() result.comparison = ComparisonInfo( diff --git a/tests/unit/test_regression_insight.py b/tests/unit/test_regression_insight.py index 0f236050..817629d8 100644 --- a/tests/unit/test_regression_insight.py +++ b/tests/unit/test_regression_insight.py @@ -26,8 +26,16 @@ def _run( evaluators: list[str] | None = None, thresholds: dict | None = None, accuracy: float = 0.9, + metrics: dict[str, float] | None = None, commit: CommitInfo | None = None, + cloud_evaluation: dict | None = None, ) -> RunResult: + aggregate_metrics = {"accuracy": accuracy} + if metrics is not None: + aggregate_metrics.update(metrics) + config: dict = {"thresholds": thresholds or {}} + if cloud_evaluation is not None: + config["cloud_evaluation"] = cloud_evaluation return RunResult( started_at="2026-09-01T10:00:00+00:00", finished_at="2026-09-01T10:00:01+00:00", @@ -41,7 +49,7 @@ def _run( ), dataset_path=dataset_path, evaluators=evaluators or ["CoherenceEvaluator"], - aggregate_metrics={"accuracy": accuracy}, + aggregate_metrics=aggregate_metrics, summary=RunSummary( items_total=1, items_passed_all=1, @@ -51,7 +59,7 @@ def _run( threshold_pass_rate=1.0, overall_passed=True, ), - config={"thresholds": thresholds or {}}, + config=config, commit=commit, ) @@ -138,14 +146,14 @@ def test_no_insight_when_from_commit_missing(): from_run = _run(accuracy=0.91, commit=None) to_run = _run(accuracy=0.79, commit=_commit("b" * 40)) - assert regression_insight.build_regression_insight(from_run, to_run, metric="accuracy") is None + assert regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) is None def test_no_insight_when_to_commit_missing(): from_run = _run(accuracy=0.91, commit=_commit("a" * 40)) to_run = _run(accuracy=0.79, commit=None) - assert regression_insight.build_regression_insight(from_run, to_run, metric="accuracy") is None + assert regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) is None def test_no_insight_when_metric_missing_on_either_run(): @@ -153,7 +161,7 @@ def test_no_insight_when_metric_missing_on_either_run(): to_run = _run(commit=_commit("b" * 40)) to_run.aggregate_metrics = {} - assert regression_insight.build_regression_insight(from_run, to_run, metric="accuracy") is None + assert regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) is None def test_insight_names_metric_before_after_and_changed_inputs(): @@ -170,12 +178,13 @@ def test_insight_names_metric_before_after_and_changed_inputs(): commit=_commit("b" * 40, short_sha="bbbbbbb"), ) - insight = regression_insight.build_regression_insight(from_run, to_run, metric="accuracy") + insight = regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) assert insight is not None - assert insight.metric == "accuracy" - assert insight.from_value == 0.91 - assert insight.to_value == 0.79 + assert len(insight.regressed_metrics) == 1 + assert insight.regressed_metrics[0].metric == "accuracy" + assert insight.regressed_metrics[0].from_value == 0.91 + assert insight.regressed_metrics[0].to_value == 0.79 assert "0.91" in insight.explanation assert "0.79" in insight.explanation assert "aaaaaaa" in insight.explanation @@ -190,7 +199,7 @@ def test_insight_explanation_has_no_cause_when_nothing_tracked_changed(): from_run = _run(accuracy=0.91, commit=_commit("a" * 40, short_sha="aaaaaaa")) to_run = _run(accuracy=0.79, commit=_commit("b" * 40, short_sha="bbbbbbb")) - insight = regression_insight.build_regression_insight(from_run, to_run, metric="accuracy") + insight = regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) assert insight is not None assert insight.changed_inputs == [] @@ -198,23 +207,170 @@ def test_insight_explanation_has_no_cause_when_nothing_tracked_changed(): assert insight.suggested_action is None -def test_used_git_diff_true_when_both_commits_locally_resolvable(monkeypatch): +def test_commits_available_locally_true_when_both_commits_locally_resolvable(monkeypatch): monkeypatch.setattr(regression_insight, "commit_exists_locally", lambda sha, **kw: True) from_run = _run(accuracy=0.91, commit=_commit("a" * 40)) to_run = _run(accuracy=0.79, commit=_commit("b" * 40)) - insight = regression_insight.build_regression_insight(from_run, to_run, metric="accuracy") + insight = regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) assert insight is not None - assert insight.used_git_diff is True + assert insight.commits_available_locally is True -def test_used_git_diff_false_when_commits_not_locally_resolvable(monkeypatch): +def test_commits_available_locally_false_when_commits_not_locally_resolvable(monkeypatch): monkeypatch.setattr(regression_insight, "commit_exists_locally", lambda sha, **kw: False) from_run = _run(accuracy=0.91, commit=_commit("a" * 40)) to_run = _run(accuracy=0.79, commit=_commit("b" * 40)) - insight = regression_insight.build_regression_insight(from_run, to_run, metric="accuracy") + insight = regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) assert insight is not None - assert insight.used_git_diff is False + assert insight.commits_available_locally is False + + +# --------------------------------------------------------------------------- +# Multiple regressed metrics (not just the worst one) +# --------------------------------------------------------------------------- + + +def test_insight_lists_all_three_regressed_metrics_ordered_worst_first(): + """similarity, coherence, and avg_latency_seconds all regress together - + every one of them must appear in `regressed_metrics`, not just the + single worst one, and ordering must be direction-aware (latency rising + is a regression, not an improvement).""" + from_run = _run( + accuracy=0.91, + metrics={"similarity": 4.0, "coherence": 4.5, "avg_latency_seconds": 2.0}, + commit=_commit("a" * 40), + ) + to_run = _run( + accuracy=0.91, + # similarity: -25%, coherence: -56%, latency: +150% (all regressions) + metrics={"similarity": 3.0, "coherence": 2.0, "avg_latency_seconds": 5.0}, + commit=_commit("b" * 40), + ) + + insight = regression_insight.build_regression_insight( + from_run, to_run, metrics=["similarity", "coherence", "avg_latency_seconds"] + ) + + assert insight is not None + assert [m.metric for m in insight.regressed_metrics] == [ + "avg_latency_seconds", + "coherence", + "similarity", + ] + assert "similarity" in insight.explanation + assert "coherence" in insight.explanation + assert "avg_latency_seconds" in insight.explanation + assert "increased from 2.00 to 5.00" in insight.explanation + + +def test_insight_skips_only_the_metric_missing_a_value_not_the_whole_insight(): + """`coherence` is missing on `to_run` - the insight should still cover + `similarity`, the metric that does have values on both sides, rather + than bailing out entirely.""" + from_run = _run( + accuracy=0.91, + metrics={"similarity": 4.0, "coherence": 4.5}, + commit=_commit("a" * 40), + ) + to_run = _run( + accuracy=0.91, + metrics={"similarity": 3.0}, + commit=_commit("b" * 40), + ) + + insight = regression_insight.build_regression_insight( + from_run, to_run, metrics=["similarity", "coherence"] + ) + + assert insight is not None + assert [m.metric for m in insight.regressed_metrics] == ["similarity"] + + +# --------------------------------------------------------------------------- +# regression_severity +# --------------------------------------------------------------------------- + + +def test_regression_severity_is_direction_aware_for_latency(): + # Latency rising from 4 to 8 is a 100% regression. + assert regression_insight.regression_severity("avg_latency_seconds", 4.0, 8.0) == 1.0 + # Latency falling is an improvement, not a regression - severity is 0. + assert regression_insight.regression_severity("avg_latency_seconds", 4.0, 2.0) == 0.0 + + +def test_regression_severity_uses_absolute_change_when_from_value_is_zero_or_negative(): + assert regression_insight.regression_severity("coherence", 0.0, -5.0) == 5.0 + assert regression_insight.regression_severity("coherence", -1.0, -2.0) == 1.0 + + +# --------------------------------------------------------------------------- +# report_url (Foundry Evaluations deep-link) +# --------------------------------------------------------------------------- + + +def test_report_urls_carried_through_when_provided(): + from_run = _run(accuracy=0.91, commit=_commit("a" * 40)) + to_run = _run(accuracy=0.79, commit=_commit("b" * 40)) + + insight = regression_insight.build_regression_insight( + from_run, + to_run, + metrics=["accuracy"], + from_report_url="https://ai.azure.com/foundry/from", + to_report_url="https://ai.azure.com/foundry/to", + ) + + assert insight is not None + assert insight.from_report_url == "https://ai.azure.com/foundry/from" + assert insight.to_report_url == "https://ai.azure.com/foundry/to" + + +def test_report_urls_are_none_when_not_provided(): + from_run = _run(accuracy=0.91, commit=_commit("a" * 40)) + to_run = _run(accuracy=0.79, commit=_commit("b" * 40)) + + insight = regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) + + assert insight is not None + assert insight.from_report_url is None + assert insight.to_report_url is None + + +def test_resolve_report_url_reads_from_run_config_first(tmp_path): + run = _run(cloud_evaluation={"report_url": "https://ai.azure.com/foundry/run1"}) + + assert ( + regression_insight.resolve_report_url(run) + == "https://ai.azure.com/foundry/run1" + ) + + +def test_resolve_report_url_falls_back_to_sidecar_file(tmp_path): + run = _run() # no cloud_evaluation in config - as for a `publish: true` local run + results_dir = tmp_path / "run1" + results_dir.mkdir() + (results_dir / "cloud_evaluation.json").write_text( + '{"report_url": "https://ai.azure.com/foundry/classic"}', encoding="utf-8" + ) + + url = regression_insight.resolve_report_url( + run, results_path=results_dir / "results.json" + ) + + assert url == "https://ai.azure.com/foundry/classic" + + +def test_resolve_report_url_is_none_when_never_published(tmp_path): + run = _run() + results_dir = tmp_path / "run1" + results_dir.mkdir() + + url = regression_insight.resolve_report_url( + run, results_path=results_dir / "results.json" + ) + + assert url is None From 0346262079305266a32f01aa5d602a243d7453d6 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 09:40:59 -0300 Subject: [PATCH 16/30] fix(cockpit): version history is direction-aware, names regressed metrics --- src/agentops/agent/cockpit.py | 127 +++++++++++++++--- tests/unit/test_cockpit.py | 234 +++++++++++++++++++++++++++++++++- 2 files changed, 341 insertions(+), 20 deletions(-) diff --git a/src/agentops/agent/cockpit.py b/src/agentops/agent/cockpit.py index 23975d1b..c8d25c76 100644 --- a/src/agentops/agent/cockpit.py +++ b/src/agentops/agent/cockpit.py @@ -39,6 +39,7 @@ summarize_redteam_readiness, ) from agentops.core.results import RunResult +from agentops.pipeline.comparison import LOWER_IS_BETTER_METRICS, metric_improved from agentops.pipeline.regression_insight import build_changed_inputs from agentops.utils.yaml import load_yaml @@ -348,7 +349,7 @@ def _build_metrics_cards(eval_runs: List[Dict[str, Any]]) -> List[Dict[str, Any] alt_links = [r.get("alt_link") for r, _v in paired] alt_labels = [r.get("alt_label") for r, _v in paired] latest = series[-1] - is_latency = key == "avg_latency_seconds" + is_latency = key in LOWER_IS_BETTER_METRICS badge = _metric_trend_badge(series, is_latency=is_latency) cards.append({ "key": key, @@ -854,7 +855,10 @@ def _build_eval_history_section(eval_runs: List[Dict[str, Any]]) -> Dict[str, An "commit_subject": commit.get("subject") if commit else None, "changed_inputs": run.get("changed_inputs") or [], "regressed": bool(run.get("regressed")), + "regressed_metrics": run.get("regressed_metrics") or [], "report_link": run.get("report_link"), + "cloud_report_url": run.get("cloud_report_url"), + "previous_cloud_report_url": run.get("previous_cloud_report_url"), } ) return {"has_runs": True, "entries": entries} @@ -927,40 +931,100 @@ def _version_lineage_key(data: Dict[str, Any]) -> Optional[str]: def _attach_version_history(runs: List[Dict[str, Any]]) -> None: - """Fill in ``changed_inputs``/``regressed`` for each run, in place. + """Fill in ``changed_inputs``/``regressed``/``previous_cloud_report_url`` + for each run, in place. Compares each run against the previous entry (in the already oldest-to-newest ordered ``runs`` list) that shares the same - ``methodology_fingerprint`` - the same grouping key Doctor's regression - check and ``results_history`` already use. A run with no fingerprint or - no prior comparable run gets an empty ``changed_inputs`` list, not a + ``version_lineage_key`` (see ``_version_lineage_key`` above) - *not* + the same thing as Doctor's rolling regression check's + ``methodology_fingerprint`` grouping (``results_history``): that one is + deliberately finer (version/deployment included), so it treats a + version bump as a new methodology and excludes it from the rolling + baseline, whereas this view groups *across* version bumps on purpose, + since showing what changed across them is the point. A run with no key + or no prior comparable run gets an empty ``changed_inputs`` list, not a fabricated one. The private ``_full_result`` helper key (a parsed ``RunResult``, not JSON-safe) is removed before returning. """ - last_by_fingerprint: Dict[str, RunResult] = {} + last_by_lineage_key: Dict[str, RunResult] = {} + last_cloud_report_url_by_lineage_key: Dict[str, Optional[str]] = {} for run in runs: - fingerprint = run.get("methodology_fingerprint") + lineage_key = run.get("version_lineage_key") current_full = cast(Optional[RunResult], run.pop("_full_result", None)) run["changed_inputs"] = [] run["regressed"] = False - - if fingerprint is not None: - previous_full = last_by_fingerprint.get(fingerprint) + run["regressed_metrics"] = [] + # The previous comparable run's own Foundry Evaluations link (its + # `cloud_report_url`, never the local-report fallback - omitted + # the same way `RegressionInsight.from_report_url` is when that + # run was never published), so a regressed row can link out to + # both sides of the comparison, not just its own. + run["previous_cloud_report_url"] = None + + if lineage_key is not None: + previous_full = last_by_lineage_key.get(lineage_key) if previous_full is not None and current_full is not None: changes = build_changed_inputs(previous_full, current_full) run["changed_inputs"] = [c.model_dump(mode="json") for c in changes] shared_metrics = set(previous_full.aggregate_metrics) & set( current_full.aggregate_metrics ) - run["regressed"] = any( - current_full.aggregate_metrics[m] < previous_full.aggregate_metrics[m] + # Named, not just a boolean, so a run that regressed on + # several metrics at once doesn't read the same as one that + # regressed on a single metric. + run["regressed_metrics"] = sorted( + m for m in shared_metrics + if metric_improved( + m, + current_full.aggregate_metrics[m], + previous_full.aggregate_metrics[m], + ) + is False + ) + run["regressed"] = bool(run["regressed_metrics"]) + run["previous_cloud_report_url"] = last_cloud_report_url_by_lineage_key.get( + lineage_key ) if current_full is not None: - last_by_fingerprint[fingerprint] = current_full + last_by_lineage_key[lineage_key] = current_full + last_cloud_report_url_by_lineage_key[lineage_key] = run.get( + "cloud_report_url" + ) + + +# Keyed by results.json path, holding (mtime, projected dict) - avoids +# re-reading, re-parsing, and re-validating the same completed run (a +# `RunResult.model_validate` over every row/metric, not cheap) on every +# cockpit render. Safe to reuse across renders because a run's +# `results.json` is written once and never mutated afterwards; `mtime` is +# still checked so a changed file (e.g. a replayed/overwritten run during +# development) isn't served stale. Each lookup returns a shallow copy so +# `_attach_version_history`'s in-place `run[...] = ...` / `run.pop(...)` +# on the result never corrupts the cached entry. +_PROJECT_RUN_CACHE: Dict[Path, Tuple[float, Optional[Dict[str, Any]]]] = {} def _project_run(path: Path, *, run_id: str) -> Optional[Dict[str, Any]]: + try: + mtime = path.stat().st_mtime + except OSError: + mtime = None + + if mtime is not None: + cached = _PROJECT_RUN_CACHE.get(path) + if cached is not None and cached[0] == mtime: + cached_projection = cached[1] + return dict(cached_projection) if cached_projection is not None else None + + projection = _project_run_uncached(path, run_id=run_id) + if mtime is not None: + _PROJECT_RUN_CACHE[path] = (mtime, dict(projection) if projection is not None else None) + return projection + + +def _project_run_uncached(path: Path, *, run_id: str) -> Optional[Dict[str, Any]]: try: data = json.loads(path.read_text(encoding="utf-8")) except (OSError, ValueError): @@ -1019,7 +1083,7 @@ def _project_run(path: Path, *, run_id: str) -> Optional[Dict[str, Any]]: "alt_link": alt_link, "alt_label": alt_label, "commit": commit if isinstance(commit, dict) else None, - "methodology_fingerprint": _version_lineage_key(data), + "version_lineage_key": _version_lineage_key(data), # Internal only - consumed and removed by `_attach_version_history`. "_full_result": full_result, } @@ -4340,9 +4404,34 @@ def _render_eval_history_section(eval_history: Dict[str, Any]) -> str: else: changes_html = 'no tracked changes' + regressed_metrics = entry.get("regressed_metrics") or [] regressed_badge = ( - 'regressed' - if entry.get("regressed") + 'regressed: ' + f'{_html_escape(", ".join(regressed_metrics))}' + if regressed_metrics + else "" + ) + # Links to both sides of the comparison in Foundry, when each was + # published there - same data as RegressionInsight.from_report_url/ + # to_report_url, omitted silently when a side was never published + # (e.g. local execution with no `publish: true`). Only shown + # alongside the regressed badge; every row already links to its + # own report via run_label regardless of whether it regressed. + foundry_links: List[str] = [] + if regressed_metrics: + previous_cloud_url = entry.get("previous_cloud_report_url") + current_cloud_url = entry.get("cloud_report_url") + if previous_cloud_url: + foundry_links.append( + f'baseline in Foundry' + ) + if current_cloud_url: + foundry_links.append( + f'this run in Foundry' + ) + foundry_links_html = ( + f' {" · ".join(foundry_links)}' + if foundry_links else "" ) report_link = entry.get("report_link") @@ -4352,7 +4441,7 @@ def _render_eval_history_section(eval_history: Dict[str, Any]) -> str: rows.append( '
' - f'
{run_label} {regressed_badge}
' + f'
{run_label} {regressed_badge}{foundry_links_html}
' f'
{timestamp} · {commit_html}
' f'
{metrics_html}
' f'
{changes_html}
' @@ -5091,6 +5180,10 @@ def _shorten_workspace(path: str) -> str: }} .history-run {{ font-size: 13px; font-weight: 600; color: var(--text); }} .history-run a {{ color: inherit; }} + .history-foundry-links {{ + font-size: 11px; font-weight: 400; color: var(--text-dim); + }} + .history-foundry-links a {{ color: var(--accent); }} .history-meta {{ font-size: 12px; color: var(--text-dim); margin-top: 2px; }} diff --git a/tests/unit/test_cockpit.py b/tests/unit/test_cockpit.py index e4eca270..503783dd 100644 --- a/tests/unit/test_cockpit.py +++ b/tests/unit/test_cockpit.py @@ -3,6 +3,7 @@ from __future__ import annotations import json +import os from datetime import datetime, timezone from pathlib import Path from types import SimpleNamespace @@ -10,6 +11,7 @@ import pytest +from agentops.agent import cockpit as cockpit_module from agentops.agent.cockpit import ( _load_eval_runs, build_cockpit_payload, @@ -1801,10 +1803,12 @@ def _write_full_eval_run( *, timestamp_dir: str, accuracy: float, + avg_latency_seconds: float | None = None, version: str = "3", deployment: str = "gpt-4o", commit_sha: str | None, started_at: str, + report_url: str | None = None, ) -> None: """Writes a ``results.json`` with every field ``RunResult`` requires. @@ -1816,6 +1820,9 @@ def _write_full_eval_run( """ out = workspace / ".agentops" / "results" / timestamp_dir out.mkdir(parents=True, exist_ok=True) + aggregate_metrics: dict = {"accuracy": accuracy} + if avg_latency_seconds is not None: + aggregate_metrics["avg_latency_seconds"] = avg_latency_seconds payload: dict = { "version": 1, "started_at": started_at, @@ -1831,7 +1838,7 @@ def _write_full_eval_run( "dataset_path": "data/smoke.jsonl", "evaluators": ["CoherenceEvaluator"], "rows": [], - "aggregate_metrics": {"accuracy": accuracy}, + "aggregate_metrics": aggregate_metrics, "thresholds": [], "summary": { "items_total": 1, @@ -1854,9 +1861,13 @@ def _write_full_eval_run( "source": "ci", } (out / "results.json").write_text(json.dumps(payload), encoding="utf-8") + if report_url is not None: + (out / "cloud_evaluation.json").write_text( + json.dumps({"report_url": report_url}), encoding="utf-8" + ) -def test_project_run_includes_commit_and_fingerprint(tmp_path: Path): +def test_project_run_includes_commit_and_lineage_key(tmp_path: Path): _write_full_eval_run( tmp_path, timestamp_dir="2026-09-01T10-00-00Z", @@ -1870,7 +1881,7 @@ def test_project_run_includes_commit_and_fingerprint(tmp_path: Path): assert len(runs) == 1 run = runs[0] assert run["commit"]["sha"] == "a" * 40 - assert run["methodology_fingerprint"] is not None + assert run["version_lineage_key"] is not None assert "_full_result" not in run assert run["changed_inputs"] == [] assert run["regressed"] is False @@ -1923,6 +1934,223 @@ def test_version_history_names_changes_vs_previous_run(tmp_path: Path): assert second["regressed"] is True +def test_version_history_latency_improvement_is_not_flagged_as_regressed( + tmp_path: Path, +): + """``avg_latency_seconds`` is lower-is-better - a drop in latency (a + real improvement) must not be reported as a regression just because the + raw number went down.""" + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-01T10-00-00Z", + accuracy=0.91, + avg_latency_seconds=8.0, + commit_sha="a" * 40, + started_at="2026-09-01T10:00:00+00:00", + ) + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-10T14-03-00Z", + accuracy=0.91, + avg_latency_seconds=3.0, + commit_sha="b" * 40, + started_at="2026-09-10T14:03:00+00:00", + ) + + runs = _load_eval_runs(tmp_path) + + assert len(runs) == 2 + _first, second = runs + assert second["regressed"] is False + + +def test_version_history_latency_regression_is_flagged(tmp_path: Path): + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-01T10-00-00Z", + accuracy=0.91, + avg_latency_seconds=3.0, + commit_sha="a" * 40, + started_at="2026-09-01T10:00:00+00:00", + ) + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-10T14-03-00Z", + accuracy=0.91, + avg_latency_seconds=8.0, + commit_sha="b" * 40, + started_at="2026-09-10T14:03:00+00:00", + ) + + runs = _load_eval_runs(tmp_path) + + assert len(runs) == 2 + _first, second = runs + assert second["regressed"] is True + assert second["regressed_metrics"] == ["avg_latency_seconds"] + + +def test_version_history_names_every_regressed_metric_not_just_a_boolean( + tmp_path: Path, +): + """accuracy and avg_latency_seconds both regress at once - both must be + named in `regressed_metrics` (and in the rendered badge), not collapsed + into a single generic "regressed" flag.""" + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-01T10-00-00Z", + accuracy=0.91, + avg_latency_seconds=3.0, + commit_sha="a" * 40, + started_at="2026-09-01T10:00:00+00:00", + ) + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-10T14-03-00Z", + accuracy=0.79, + avg_latency_seconds=8.0, + commit_sha="b" * 40, + started_at="2026-09-10T14:03:00+00:00", + ) + + runs = _load_eval_runs(tmp_path) + + assert len(runs) == 2 + _first, second = runs + assert second["regressed_metrics"] == ["accuracy", "avg_latency_seconds"] + + payload = build_cockpit_payload(tmp_path) + html = render_cockpit_html(payload) + assert "regressed: accuracy, avg_latency_seconds" in html + + +def test_version_history_links_both_sides_to_foundry_when_both_published( + tmp_path: Path, +): + """A regressed row must link out to both the baseline's and its own + Foundry Evaluations page - the same data as + ``RegressionInsight.from_report_url``/``to_report_url``, just surfaced + in Cockpit instead of report.md.""" + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-01T10-00-00Z", + accuracy=0.91, + commit_sha="a" * 40, + started_at="2026-09-01T10:00:00+00:00", + report_url="https://ai.azure.com/foundry/baseline", + ) + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-10T14-03-00Z", + accuracy=0.79, + commit_sha="b" * 40, + started_at="2026-09-10T14:03:00+00:00", + report_url="https://ai.azure.com/foundry/current", + ) + + runs = _load_eval_runs(tmp_path) + assert len(runs) == 2 + _first, second = runs + assert second["previous_cloud_report_url"].startswith( + "https://ai.azure.com/foundry/baseline" + ) + assert second["cloud_report_url"].startswith("https://ai.azure.com/foundry/current") + + payload = build_cockpit_payload(tmp_path) + html = render_cockpit_html(payload) + assert "baseline in Foundry" in html + assert "this run in Foundry" in html + + +def test_version_history_omits_foundry_links_when_not_published(tmp_path: Path): + """Neither run was published (no ``cloud_evaluation.json`` sidecar) - + the Foundry links must be omitted silently, not shown as broken links + or placeholders.""" + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-01T10-00-00Z", + accuracy=0.91, + commit_sha="a" * 40, + started_at="2026-09-01T10:00:00+00:00", + ) + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-10T14-03-00Z", + accuracy=0.79, + commit_sha="b" * 40, + started_at="2026-09-10T14:03:00+00:00", + ) + + runs = _load_eval_runs(tmp_path) + assert len(runs) == 2 + _first, second = runs + assert second["regressed"] is True + assert second["previous_cloud_report_url"] is None + assert second["cloud_report_url"] is None + + payload = build_cockpit_payload(tmp_path) + html = render_cockpit_html(payload) + assert "baseline in Foundry" not in html + assert "this run in Foundry" not in html + + +def test_project_run_is_cached_by_path_and_mtime(tmp_path: Path, monkeypatch): + """Re-rendering the cockpit without any new run must not re-parse and + re-validate (``RunResult.model_validate``) the same unchanged + ``results.json`` files again - that cost is paid once per file, not + once per render.""" + cockpit_module._PROJECT_RUN_CACHE.clear() + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-01T10-00-00Z", + accuracy=0.91, + commit_sha="a" * 40, + started_at="2026-09-01T10:00:00+00:00", + ) + + calls = {"count": 0} + original = cockpit_module._project_run_uncached + + def _counting_uncached(path, *, run_id): + calls["count"] += 1 + return original(path, run_id=run_id) + + monkeypatch.setattr(cockpit_module, "_project_run_uncached", _counting_uncached) + + first = _load_eval_runs(tmp_path) + second = _load_eval_runs(tmp_path) + + assert calls["count"] == 1 + assert first[0]["run_id"] == second[0]["run_id"] + assert first[0]["commit"] == second[0]["commit"] + + +def test_project_run_cache_invalidated_on_file_change(tmp_path: Path, monkeypatch): + cockpit_module._PROJECT_RUN_CACHE.clear() + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-01T10-00-00Z", + accuracy=0.91, + commit_sha="a" * 40, + started_at="2026-09-01T10:00:00+00:00", + ) + + first = _load_eval_runs(tmp_path) + assert first[0]["metrics"]["accuracy"] == 0.91 + + results_path = tmp_path / ".agentops" / "results" / "2026-09-01T10-00-00Z" / "results.json" + payload = json.loads(results_path.read_text(encoding="utf-8")) + payload["aggregate_metrics"]["accuracy"] = 0.42 + results_path.write_text(json.dumps(payload), encoding="utf-8") + # Force a distinct mtime even on filesystems with coarse mtime + # resolution, so the cache is guaranteed to observe the change. + new_mtime = results_path.stat().st_mtime + 1 + os.utime(results_path, (new_mtime, new_mtime)) + + second = _load_eval_runs(tmp_path) + assert second[0]["metrics"]["accuracy"] == 0.42 + + def test_cockpit_html_renders_version_history_section(tmp_path: Path): _write_full_eval_run( tmp_path, From 6f8882331241da5bc76a0604cb0d6c22861752c5 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 09:41:18 -0300 Subject: [PATCH 17/30] fix(doctor): explain unavailable attribution; attribute across version bumps --- src/agentops/agent/analyzer.py | 4 +- src/agentops/agent/checks/catalog.py | 7 +- src/agentops/agent/checks/regression.py | 86 ++++++++- src/agentops/agent/sources/results_history.py | 54 ++++++ tests/unit/test_agent_checks_regression.py | 165 +++++++++++++++++- 5 files changed, 304 insertions(+), 12 deletions(-) diff --git a/src/agentops/agent/analyzer.py b/src/agentops/agent/analyzer.py index 03e527ea..be744b3f 100644 --- a/src/agentops/agent/analyzer.py +++ b/src/agentops/agent/analyzer.py @@ -159,7 +159,9 @@ def analyze( posture_config = posture_config.model_copy(update={"exclude_rules": merged}) findings: List[Finding] = [] - findings.extend(run_regression_check(history, config.checks.regression)) + findings.extend( + run_regression_check(history, config.checks.regression, workspace=workspace) + ) findings.extend(run_latency_check(history, monitor, config.checks.latency)) findings.extend(run_errors_check(monitor, foundry, config.checks.errors)) findings.extend( diff --git a/src/agentops/agent/checks/catalog.py b/src/agentops/agent/checks/catalog.py index 6e721d50..4cbce9a3 100644 --- a/src/agentops/agent/checks/catalog.py +++ b/src/agentops/agent/checks/catalog.py @@ -196,7 +196,12 @@ def is_llm_judged(self) -> bool: summary=( "For each metric in the regression watchlist, compare the " "latest run to a rolling baseline of previous runs and flag " - "drops that exceed the configured tolerance." + "drops that exceed the configured tolerance. When both runs " + "have commit metadata, the recommendation also explains what " + "changed between them; when one side is a Foundry cloud run " + "fetched as a fallback (no local results.json, no commit), the " + "recommendation says attribution is unavailable instead of " + "omitting it silently." ), severities=(Severity.WARNING, Severity.CRITICAL), requires=("results_history",), diff --git a/src/agentops/agent/checks/regression.py b/src/agentops/agent/checks/regression.py index 687db698..38cca2b6 100644 --- a/src/agentops/agent/checks/regression.py +++ b/src/agentops/agent/checks/regression.py @@ -3,6 +3,7 @@ from __future__ import annotations import json +from pathlib import Path from statistics import mean from typing import List, Optional @@ -10,7 +11,7 @@ from agentops.agent.findings import Category, Finding, Severity from agentops.agent.sources.results_history import ResultsHistory, RunSummary from agentops.core.results import RegressionInsight, RunResult -from agentops.pipeline.regression_insight import build_regression_insight +from agentops.pipeline.regression_insight import build_regression_insight, resolve_report_url def _load_run_result(summary: RunSummary) -> Optional[RunResult]: @@ -34,8 +35,46 @@ def _load_run_result(summary: RunSummary) -> Optional[RunResult]: return None +def _attribution_unavailable_reason( + latest: RunSummary, + previous_run: Optional[RunSummary], + latest_result: Optional[RunResult], + previous_result: Optional[RunResult], +) -> str: + """Why ``build_regression_insight`` was skipped or returned ``None``. + + Surfaced instead of silently falling back to the generic + recommendation, since a cloud-fetched run (Doctor's fallback to + Foundry evaluation runs when local history is too short - see + ``results_history._merge_runs``) has no local ``results.json`` to + reload and therefore no recorded commit, unlike every run produced by + ``agentops eval run`` (including ``execution: cloud``/``azd``), which + always attempts commit capture. + """ + if previous_run is None: + return "attribution unavailable: no comparable prior run found" + non_local = [ + run + for run, result in ((latest, latest_result), (previous_run, previous_result)) + if result is None and run.source != "local" + ] + if non_local: + return "attribution unavailable: run has no commit (fetched from cloud)" + if latest_result is None or previous_result is None: + return "attribution unavailable: the run's results.json could not be read" + if latest_result.commit is None or previous_result.commit is None: + return "attribution unavailable: run has no recorded commit" + return ( + "attribution unavailable: the regressed metric's value is missing " + "on one of the compared runs" + ) + + def run_regression_check( - history: ResultsHistory, config: RegressionCheckConfig + history: ResultsHistory, + config: RegressionCheckConfig, + *, + workspace: Optional[Path] = None, ) -> List[Finding]: runs = history.runs if len(runs) < config.min_runs: @@ -58,12 +97,23 @@ def run_regression_check( if not baseline_runs: return [] - # The immediately preceding comparable run - the natural "what changed - # since last time" pair, distinct from the rolling mean used for the - # drop-percentage math below. - previous_run = baseline_runs[-1] + # The immediately preceding comparable run, for causal attribution - + # distinct from `baseline_runs` above (used for the rolling drop% + # mean). Deliberately keyed on the coarser `lineage_key` (dataset, + # evaluators, agent identity - version/deployment excluded) rather than + # `baseline_runs`/`methodology_fingerprint`: a version bump changes the + # fingerprint, so using `baseline_runs` here would always filter out + # the very run (the old version) this feature's canonical scenario - + # attributing a regression to a version bump - needs to diff against. + lineage_key = latest.lineage_key + if lineage_key is None: + lineage_runs = runs[:-1] + else: + lineage_runs = [r for r in runs[:-1] if r.lineage_key == lineage_key] + previous_run = lineage_runs[-1] if lineage_runs else None + latest_result = _load_run_result(latest) - previous_result = _load_run_result(previous_run) + previous_result = _load_run_result(previous_run) if previous_run is not None else None findings: List[Finding] = [] for metric in config.metrics: @@ -107,12 +157,32 @@ def run_regression_check( insight: Optional[RegressionInsight] = None if latest_result is not None and previous_result is not None: - insight = build_regression_insight(previous_result, latest_result, metric=metric) + insight = build_regression_insight( + previous_result, + latest_result, + metrics=[metric], + workspace=workspace, + # Both are already-completed historical runs, so a sidecar + # `cloud_evaluation.json` next to either (from + # `execution: cloud` or a finished `publish: true`) is + # complete by now - unlike the in-flight `--baseline` + # comparison in `pipeline.comparison`. + from_report_url=resolve_report_url( + previous_result, results_path=previous_run.raw_path + ), + to_report_url=resolve_report_url(latest_result, results_path=latest.raw_path), + ) if insight is not None: evidence["insight"] = insight.model_dump(mode="json") recommendation = insight.explanation if insight.suggested_action: recommendation = f"{recommendation} {insight.suggested_action}" + else: + reason = _attribution_unavailable_reason( + latest, previous_run, latest_result, previous_result + ) + evidence["attribution_unavailable"] = reason + recommendation = f"{recommendation} ({reason})" findings.append( Finding( diff --git a/src/agentops/agent/sources/results_history.py b/src/agentops/agent/sources/results_history.py index ad53b688..8773855a 100644 --- a/src/agentops/agent/sources/results_history.py +++ b/src/agentops/agent/sources/results_history.py @@ -36,6 +36,11 @@ class RunSummary: source: str = "local" portal_url: Optional[str] = None methodology_fingerprint: Optional[str] = None + # Coarser than `methodology_fingerprint` - same agent identity, dataset, + # and evaluator set, but version/deployment excluded. See + # `_lineage_key`'s docstring for why `agent.checks.regression` needs + # this separate, coarser key for causal attribution. + lineage_key: Optional[str] = None @dataclass @@ -139,9 +144,58 @@ def _summarize(path: Path) -> Optional[RunSummary]: raw_path=path, item_evaluations=item_evaluations, methodology_fingerprint=_methodology_fingerprint(data), + lineage_key=_lineage_key(data), ) +def _lineage_key(data: Dict[str, Any]) -> Optional[str]: + """Derive a stable hash of (agent identity, dataset, evaluators) - + deliberately excluding the agent's version/deployment, unlike + ``_methodology_fingerprint`` below. + + ``agent.checks.regression`` needs this coarser key to pick "the + immediately preceding comparable run" for causal regression + attribution. Using ``_methodology_fingerprint`` for that (as it + correctly does for its own drop%-mean baseline, where mixing + methodologies would be spurious) would make the feature's own + canonical scenario - attributing a regression to a version bump, + e.g. "run v3 -> v4" - unreachable: a version bump changes the + fingerprint, so the prior (different-version) run would always be + filtered out before it could be picked as the comparison partner. + """ + raw_target = data.get("target") + target: Dict[str, Any] = raw_target if isinstance(raw_target, dict) else {} + agent_identity = ( + target.get("name") + or target.get("url") + or target.get("raw") + or (data.get("config") or {}).get("agent") + ) + dataset_path = data.get("dataset_path") or (data.get("config") or {}).get( + "dataset" + ) + evaluators_raw = data.get("evaluators") + if isinstance(evaluators_raw, list): + evaluators = sorted(str(e) for e in evaluators_raw) + elif isinstance(evaluators_raw, dict): + evaluators = sorted(str(k) for k in evaluators_raw.keys()) + else: + evaluators = [] + + if not agent_identity and not dataset_path and not evaluators: + return None + + payload = json.dumps( + { + "agent_identity": str(agent_identity) if agent_identity else None, + "dataset": str(dataset_path) if dataset_path else None, + "evaluators": evaluators, + }, + sort_keys=True, + ) + return hashlib.sha256(payload.encode("utf-8")).hexdigest()[:16] + + def _methodology_fingerprint(data: Dict[str, Any]) -> Optional[str]: """Derive a stable hash of (agent target, dataset, evaluators). diff --git a/tests/unit/test_agent_checks_regression.py b/tests/unit/test_agent_checks_regression.py index 0bb03424..edfe2f83 100644 --- a/tests/unit/test_agent_checks_regression.py +++ b/tests/unit/test_agent_checks_regression.py @@ -17,6 +17,7 @@ def _run( run_id: str = "r", offset_days: int = 0, fingerprint: str | None = None, + lineage_key: str | None = None, raw_path: Path | None = None, source: str = "local", ) -> RunSummary: @@ -29,6 +30,7 @@ def _run( items_passed_all=1, raw_path=raw_path or Path("dummy"), methodology_fingerprint=fingerprint, + lineage_key=lineage_key, source=source, ) @@ -182,11 +184,46 @@ def test_regression_check_attaches_insight_when_both_runs_have_commit_metadata(t assert len(findings) == 1 finding = findings[0] assert "insight" in finding.evidence - assert finding.evidence["insight"]["metric"] == "accuracy" + assert finding.evidence["insight"]["regressed_metrics"][0]["metric"] == "accuracy" assert "gpt-4o" in finding.recommendation and "gpt-4o-mini" in finding.recommendation assert "prompt" in finding.recommendation +def test_regression_check_insight_carries_report_url_when_published_present_and_absent( + tmp_path, +) -> None: + """The baseline run was published (sidecar ``cloud_evaluation.json`` + next to its ``results.json``); the latest run was not - the insight + must carry the former's URL and omit the latter's, rather than + fabricating or erroring on the missing side.""" + baseline_path = tmp_path / "baseline" / "results.json" + latest_path = tmp_path / "latest" / "results.json" + _write_result_json( + baseline_path, accuracy=0.91, version="3", deployment="gpt-4o", commit_sha="a" * 40 + ) + (baseline_path.parent / "cloud_evaluation.json").write_text( + '{"report_url": "https://ai.azure.com/foundry/baseline"}', encoding="utf-8" + ) + _write_result_json( + latest_path, accuracy=0.79, version="4", deployment="gpt-4o-mini", commit_sha="b" * 40 + ) + history = ResultsHistory( + runs=[ + _run({"accuracy": 0.91}, run_id="b1", offset_days=-3, raw_path=baseline_path), + _run({"accuracy": 0.91}, run_id="b2", offset_days=-2, raw_path=baseline_path), + _run({"accuracy": 0.79}, run_id="latest", offset_days=0, raw_path=latest_path), + ] + ) + config = RegressionCheckConfig(metrics=["accuracy"], threshold_drop=0.10, min_runs=3) + + findings = run_regression_check(history, config) + + assert len(findings) == 1 + insight = findings[0].evidence["insight"] + assert insight["from_report_url"] == "https://ai.azure.com/foundry/baseline" + assert insight["to_report_url"] is None + + def test_regression_check_keeps_generic_recommendation_without_commit_metadata() -> None: """Today's behavior is unchanged when raw_path can't be reloaded (e.g. missing file).""" history = ResultsHistory( @@ -205,7 +242,9 @@ def test_regression_check_keeps_generic_recommendation_without_commit_metadata() assert "inspect prompt/model/dataset changes" in findings[0].recommendation -def test_regression_check_skips_insight_for_cloud_sourced_runs(tmp_path) -> None: +def test_regression_check_explains_why_attribution_is_unavailable_for_cloud_sourced_runs( + tmp_path, +) -> None: baseline_path = tmp_path / "baseline" / "results.json" latest_path = tmp_path / "latest" / "results.json" _write_result_json( @@ -245,3 +284,125 @@ def test_regression_check_skips_insight_for_cloud_sourced_runs(tmp_path) -> None assert len(findings) == 1 assert "insight" not in findings[0].evidence + assert findings[0].evidence["attribution_unavailable"] == ( + "attribution unavailable: run has no commit (fetched from cloud)" + ) + assert "attribution unavailable: run has no commit (fetched from cloud)" in ( + findings[0].recommendation + ) + + +def test_regression_check_explains_unavailable_attribution_for_mixed_local_and_cloud_runs( + tmp_path, +) -> None: + """A local baseline compared against a cloud-fetched current run (the + asymmetric case Doctor's history merge can produce) - no commit is + available on the cloud side, so attribution is unavailable, and the + finding must say why rather than silently falling back to the generic + recommendation.""" + baseline_path = tmp_path / "baseline" / "results.json" + _write_result_json( + baseline_path, accuracy=0.91, version="3", deployment="gpt-4o", commit_sha="a" * 40 + ) + history = ResultsHistory( + runs=[ + _run( + {"accuracy": 0.91}, + run_id="b1", + offset_days=-3, + raw_path=baseline_path, + source="local", + ), + _run( + {"accuracy": 0.91}, + run_id="b2", + offset_days=-2, + raw_path=baseline_path, + source="local", + ), + _run( + {"accuracy": 0.79}, + run_id="latest", + offset_days=0, + raw_path=Path("foundry") / "eval123" / "run456", + source="foundry_cloud", + ), + ] + ) + config = RegressionCheckConfig(metrics=["accuracy"], threshold_drop=0.10, min_runs=3) + + findings = run_regression_check(history, config) + + assert len(findings) == 1 + assert "insight" not in findings[0].evidence + assert findings[0].evidence["attribution_unavailable"] == ( + "attribution unavailable: run has no commit (fetched from cloud)" + ) + + +def test_attribution_uses_the_true_immediately_preceding_run_across_a_version_change( + tmp_path, +) -> None: + """A one-off run on a different version (v5) sits between a v4 run and + the latest v4 run. `methodology_fingerprint` (version-inclusive) + excludes that v5 run from `baseline_runs`, so picking attribution's + comparison partner from `baseline_runs` (the old behavior) would + silently skip over it and compare against the older v4 run instead - + hiding the fact that something changed in between. Attribution must + use the coarser `lineage_key` instead, so it diffs against the run + that's actually immediately before `latest`. + """ + v4_path = tmp_path / "v4" / "results.json" + v5_path = tmp_path / "v5" / "results.json" + latest_path = tmp_path / "latest" / "results.json" + _write_result_json( + v4_path, accuracy=0.90, version="4", deployment="gpt-4o", commit_sha="1" * 40 + ) + _write_result_json( + v5_path, accuracy=0.95, version="5", deployment="gpt-4o", commit_sha="2" * 40 + ) + _write_result_json( + latest_path, accuracy=0.70, version="4", deployment="gpt-4o", commit_sha="3" * 40 + ) + + history = ResultsHistory( + runs=[ + _run( + {"accuracy": 0.90}, + run_id="v4", + offset_days=-2, + fingerprint="V4", + lineage_key="L", + raw_path=v4_path, + ), + _run( + {"accuracy": 0.95}, + run_id="v5", + offset_days=-1, + fingerprint="V5", + lineage_key="L", + raw_path=v5_path, + ), + _run( + {"accuracy": 0.70}, + run_id="latest", + offset_days=0, + fingerprint="V4", + lineage_key="L", + raw_path=latest_path, + ), + ] + ) + # min_runs=2 so the fingerprint-gated `baseline_runs` (just the v4 run) + # already satisfies `len(baseline_runs) + 1 >= min_runs` on its own. + config = RegressionCheckConfig(metrics=["accuracy"], threshold_drop=0.10, min_runs=2) + + findings = run_regression_check(history, config) + + assert len(findings) == 1 + insight = findings[0].evidence["insight"] + # The v5 run's commit, not the older v4 run's - confirms attribution + # diffed against the true immediately-preceding run. + assert insight["from_commit"]["sha"] == "2" * 40 + fields = {c["field"] for c in insight["changed_inputs"]} + assert "system_prompt" in fields From f0c2a354b9fdcc838ac0e8aa5c7ba0c121f6f295 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 10:16:22 -0300 Subject: [PATCH 18/30] fix(doctor,cockpit): fix lineage key for direct-model targets; detect --- src/agentops/agent/checks/regression.py | 47 +++---- src/agentops/agent/cockpit.py | 91 ++++++++++--- src/agentops/agent/sources/results_history.py | 13 +- tests/unit/test_agent_checks_regression.py | 100 +++++++++++--- tests/unit/test_agent_results_history.py | 29 +++- tests/unit/test_cockpit.py | 127 +++++++++++++++++- 6 files changed, 329 insertions(+), 78 deletions(-) diff --git a/src/agentops/agent/checks/regression.py b/src/agentops/agent/checks/regression.py index 38cca2b6..89cf0b14 100644 --- a/src/agentops/agent/checks/regression.py +++ b/src/agentops/agent/checks/regression.py @@ -37,7 +37,7 @@ def _load_run_result(summary: RunSummary) -> Optional[RunResult]: def _attribution_unavailable_reason( latest: RunSummary, - previous_run: Optional[RunSummary], + previous_run: RunSummary, latest_result: Optional[RunResult], previous_result: Optional[RunResult], ) -> str: @@ -51,8 +51,6 @@ def _attribution_unavailable_reason( ``agentops eval run`` (including ``execution: cloud``/``azd``), which always attempts commit capture. """ - if previous_run is None: - return "attribution unavailable: no comparable prior run found" non_local = [ run for run, result in ((latest, latest_result), (previous_run, previous_result)) @@ -81,39 +79,34 @@ def run_regression_check( return [] latest = runs[-1] - # Only compare against runs that share the same evaluation methodology - # (same agent target, dataset, and evaluator set). This avoids spurious - # regressions when the dataset, evaluators, or runner changes between - # runs (e.g. smoke → hardened conversation rubric, or local → cloud). - fingerprint = latest.methodology_fingerprint - if fingerprint is None: + # Only compare against runs that share the same dataset, evaluators, + # and agent identity - but deliberately version/deployment-*blind* + # (`lineage_key`, not the finer `methodology_fingerprint` opex.py's + # flaky-metric check needs, which does want version included since + # mixing versions there would inflate variance spuriously). A version + # bump is exactly the kind of change this check - and its causal + # attribution - exists to catch, not an incompatible methodology to + # exclude: keying this on the fingerprint instead would make the + # check structurally unable to fire on the first run(s) after a + # version bump (there being no baseline yet sharing that new, + # not-yet-established fingerprint), making the feature's own + # canonical "run v3 -> v4" scenario unreachable. + lineage_key = latest.lineage_key + if lineage_key is None: baseline_runs = runs[:-1] else: - baseline_runs = [ - r for r in runs[:-1] if r.methodology_fingerprint == fingerprint - ] + baseline_runs = [r for r in runs[:-1] if r.lineage_key == lineage_key] if len(baseline_runs) + 1 < config.min_runs: return [] if not baseline_runs: return [] # The immediately preceding comparable run, for causal attribution - - # distinct from `baseline_runs` above (used for the rolling drop% - # mean). Deliberately keyed on the coarser `lineage_key` (dataset, - # evaluators, agent identity - version/deployment excluded) rather than - # `baseline_runs`/`methodology_fingerprint`: a version bump changes the - # fingerprint, so using `baseline_runs` here would always filter out - # the very run (the old version) this feature's canonical scenario - - # attributing a regression to a version bump - needs to diff against. - lineage_key = latest.lineage_key - if lineage_key is None: - lineage_runs = runs[:-1] - else: - lineage_runs = [r for r in runs[:-1] if r.lineage_key == lineage_key] - previous_run = lineage_runs[-1] if lineage_runs else None - + # distinct from the rolling drop% mean below, which uses every + # comparable run, not just the last one. + previous_run = baseline_runs[-1] latest_result = _load_run_result(latest) - previous_result = _load_run_result(previous_run) if previous_run is not None else None + previous_result = _load_run_result(previous_run) findings: List[Finding] = [] for metric in config.metrics: diff --git a/src/agentops/agent/cockpit.py b/src/agentops/agent/cockpit.py index c8d25c76..d95b4f10 100644 --- a/src/agentops/agent/cockpit.py +++ b/src/agentops/agent/cockpit.py @@ -83,8 +83,17 @@ def build_cockpit_payload( *, history: Optional[List[AnalysisRecord]] = None, time_range: Optional[TimeRange] = None, + run_cache: Optional["_RunCache"] = None, ) -> Dict[str, Any]: - """Reduce local configuration and the latest Doctor run for the cockpit.""" + """Reduce local configuration and the latest Doctor run for the cockpit. + + ``run_cache`` (see ``_project_run``) is an explicit, caller-owned dict - + passing one lets repeated calls within the same long-lived process + (e.g. ``create_app``'s per-server cache) skip re-parsing and + re-validating unchanged ``results.json`` files; omitting it (the + default) does no caching at all, since a bare module-level cache would + be hidden global state the project's architecture explicitly forbids. + """ _ = time_range # Retained for callers; the cockpit no longer filters by date. records = history if history is not None else load_analysis_history(workspace) telemetry = _telemetry_status() @@ -96,7 +105,9 @@ def build_cockpit_payload( watchdog_payload, readiness, initialized=(workspace / "agentops.yaml").exists(), ) - eval_history = _build_eval_history_section(_load_eval_runs(workspace)) + eval_history = _build_eval_history_section( + _load_eval_runs(workspace, run_cache=run_cache) + ) return { "workspace": str(workspace.resolve()), @@ -864,10 +875,13 @@ def _build_eval_history_section(eval_runs: List[Dict[str, Any]]) -> Dict[str, An return {"has_runs": True, "entries": entries} -def _load_eval_runs(workspace: Path, *, limit: int = 24) -> List[Dict[str, Any]]: +def _load_eval_runs( + workspace: Path, *, limit: int = 24, run_cache: Optional["_RunCache"] = None +) -> List[Dict[str, Any]]: """Scan ``.agentops/results//results.json`` and project the fields the cockpit cares about. ``latest/`` is skipped because it is - a mirror of the most recent timestamped run. + a mirror of the most recent timestamped run. See ``_project_run`` for + ``run_cache``. """ results_root = workspace / ".agentops" / "results" if not results_root.exists(): @@ -887,7 +901,7 @@ def _load_eval_runs(workspace: Path, *, limit: int = 24) -> List[Dict[str, Any]] runs: List[Dict[str, Any]] = [] for run_id, path in candidates: - run = _project_run(path, run_id=run_id) + run = _project_run(path, run_id=run_id, run_cache=run_cache) if run is not None: runs.append(run) @@ -911,7 +925,22 @@ def _version_lineage_key(data: Dict[str, Any]) -> Optional[str]: """ raw_target = data.get("target") target: Dict[str, Any] = raw_target if isinstance(raw_target, dict) else {} - agent_identity = target.get("name") or target.get("url") or target.get("raw") + # `raw` is the identity fallback for every target kind except + # `model_direct` (`model:`), where `raw` itself embeds the + # deployment (e.g. "model:gpt-4o" vs "model:gpt-4o-mini") - using it + # there would make a deployment change start a new lineage instead of + # being detected as a change within the same one, defeating the + # version/deployment-blind point of this key. `kind` has no such + # problem: it's the constant "model_direct" for every run of this kind. + agent_identity = ( + target.get("name") + or target.get("url") + or ( + target.get("kind") + if target.get("kind") == "model_direct" + else target.get("raw") + ) + ) dataset_path = data.get("dataset_path") evaluators_raw = data.get("evaluators") evaluators = ( @@ -994,33 +1023,47 @@ def _attach_version_history(runs: List[Dict[str, Any]]) -> None: ) -# Keyed by results.json path, holding (mtime, projected dict) - avoids -# re-reading, re-parsing, and re-validating the same completed run (a -# `RunResult.model_validate` over every row/metric, not cheap) on every -# cockpit render. Safe to reuse across renders because a run's -# `results.json` is written once and never mutated afterwards; `mtime` is -# still checked so a changed file (e.g. a replayed/overwritten run during -# development) isn't served stale. Each lookup returns a shallow copy so -# `_attach_version_history`'s in-place `run[...] = ...` / `run.pop(...)` -# on the result never corrupts the cached entry. -_PROJECT_RUN_CACHE: Dict[Path, Tuple[float, Optional[Dict[str, Any]]]] = {} - +# Keyed by results.json path, holding (mtime, projected dict). Deliberately +# *not* a module-level variable (the architecture constitution and AGENTS.md +# both prohibit new hidden global state) - callers that want the caching +# benefit own and pass in their own dict (``create_app`` keeps one per +# running Cockpit server instance, scoped to its own closure); a caller +# that passes ``None`` (the default) just gets no caching, not an error. +_RunCache = Dict[Path, Tuple[float, Optional[Dict[str, Any]]]] + + +def _project_run( + path: Path, *, run_id: str, run_cache: Optional[_RunCache] = None +) -> Optional[Dict[str, Any]]: + """Project one ``results.json`` into the fields the cockpit cares about. + + ``run_cache`` avoids re-reading, re-parsing, and re-validating the same + completed run (a ``RunResult.model_validate`` over every row/metric, + not cheap) on every cockpit render. Safe to reuse across renders + because a run's ``results.json`` is written once and never mutated + afterwards; ``mtime`` is still checked so a changed file (e.g. a + replayed/overwritten run during development) isn't served stale. Each + lookup returns a shallow copy so ``_attach_version_history``'s + in-place ``run[...] = ...`` / ``run.pop(...)`` on the result never + corrupts the cached entry. + """ + if run_cache is None: + return _project_run_uncached(path, run_id=run_id) -def _project_run(path: Path, *, run_id: str) -> Optional[Dict[str, Any]]: try: mtime = path.stat().st_mtime except OSError: mtime = None if mtime is not None: - cached = _PROJECT_RUN_CACHE.get(path) + cached = run_cache.get(path) if cached is not None and cached[0] == mtime: cached_projection = cached[1] return dict(cached_projection) if cached_projection is not None else None projection = _project_run_uncached(path, run_id=run_id) if mtime is not None: - _PROJECT_RUN_CACHE[path] = (mtime, dict(projection) if projection is not None else None) + run_cache[path] = (mtime, dict(projection) if projection is not None else None) return projection @@ -5703,12 +5746,16 @@ def create_app(workspace: Path): redoc_url=None, openapi_url=None, ) + # Explicit, app-instance-scoped (not module-level/global) cache for + # `_project_run` - lives exactly as long as this server process. See + # `_project_run`'s docstring. + run_cache: _RunCache = {} @app.get("/", response_class=HTMLResponse) def _index(partial: Optional[str] = Query(None, alias="_partial")): if not partial: return HTMLResponse(_render_loading_shell()) - payload = build_cockpit_payload(workspace) + payload = build_cockpit_payload(workspace, run_cache=run_cache) return HTMLResponse(render_cockpit_html(payload)) @app.get("/favicon.ico") @@ -5728,7 +5775,7 @@ def _api_history(limit: Optional[int] = None): @app.get("/api/eval-runs") def _api_eval_runs(limit: int = 24): - return JSONResponse(_load_eval_runs(workspace, limit=limit)) + return JSONResponse(_load_eval_runs(workspace, limit=limit, run_cache=run_cache)) @app.get("/api/runs/{run_id}/report", response_class=HTMLResponse) def _api_run_report(run_id: str): diff --git a/src/agentops/agent/sources/results_history.py b/src/agentops/agent/sources/results_history.py index 8773855a..babe1179 100644 --- a/src/agentops/agent/sources/results_history.py +++ b/src/agentops/agent/sources/results_history.py @@ -165,10 +165,21 @@ def _lineage_key(data: Dict[str, Any]) -> Optional[str]: """ raw_target = data.get("target") target: Dict[str, Any] = raw_target if isinstance(raw_target, dict) else {} + # `raw` is the identity fallback for every target kind except + # `model_direct` (`model:`), where `raw` itself embeds the + # deployment (e.g. "model:gpt-4o" vs "model:gpt-4o-mini") - using it + # there would make a deployment change look like a different agent + # entirely, defeating the version/deployment-blind point of this key. + # `kind` has no such problem: it's the constant "model_direct" for + # every run of this kind. agent_identity = ( target.get("name") or target.get("url") - or target.get("raw") + or ( + target.get("kind") + if target.get("kind") == "model_direct" + else target.get("raw") + ) or (data.get("config") or {}).get("agent") ) dataset_path = data.get("dataset_path") or (data.get("config") or {}).get( diff --git a/tests/unit/test_agent_checks_regression.py b/tests/unit/test_agent_checks_regression.py index edfe2f83..2b02b1c4 100644 --- a/tests/unit/test_agent_checks_regression.py +++ b/tests/unit/test_agent_checks_regression.py @@ -125,15 +125,18 @@ def test_regression_check_skips_when_baseline_too_small() -> None: assert findings == [] -def test_regression_check_ignores_baselines_with_mismatched_methodology() -> None: - """Baselines from a different dataset/evaluator set must not count.""" +def test_regression_check_ignores_baselines_with_mismatched_lineage() -> None: + """Baselines from a different dataset/evaluator set (lineage) must not + count - unlike a version/deployment difference, which must (see + ``test_attribution_uses_the_true_immediately_preceding_run_across_a_version_change`` + below).""" history = ResultsHistory( runs=[ # These baselines used a different methodology (e.g. smoke dataset) # and must be excluded from the comparison. - _run({"coherence": 4.5}, run_id="b1", offset_days=-3, fingerprint="A"), - _run({"coherence": 4.5}, run_id="b2", offset_days=-2, fingerprint="A"), - _run({"coherence": 3.0}, run_id="latest", offset_days=0, fingerprint="B"), + _run({"coherence": 4.5}, run_id="b1", offset_days=-3, lineage_key="A"), + _run({"coherence": 4.5}, run_id="b2", offset_days=-2, lineage_key="A"), + _run({"coherence": 3.0}, run_id="latest", offset_days=0, lineage_key="B"), ] ) config = RegressionCheckConfig( @@ -143,14 +146,14 @@ def test_regression_check_ignores_baselines_with_mismatched_methodology() -> Non assert findings == [] -def test_regression_check_uses_matching_methodology_baselines() -> None: - """Baselines with the same fingerprint as the latest run drive the check.""" +def test_regression_check_uses_matching_lineage_baselines() -> None: + """Baselines with the same lineage key as the latest run drive the check.""" history = ResultsHistory( runs=[ - _run({"coherence": 4.5}, run_id="other", offset_days=-4, fingerprint="A"), - _run({"coherence": 4.5}, run_id="b1", offset_days=-3, fingerprint="B"), - _run({"coherence": 4.5}, run_id="b2", offset_days=-2, fingerprint="B"), - _run({"coherence": 3.0}, run_id="latest", offset_days=0, fingerprint="B"), + _run({"coherence": 4.5}, run_id="other", offset_days=-4, lineage_key="A"), + _run({"coherence": 4.5}, run_id="b1", offset_days=-3, lineage_key="B"), + _run({"coherence": 4.5}, run_id="b2", offset_days=-2, lineage_key="B"), + _run({"coherence": 3.0}, run_id="latest", offset_days=0, lineage_key="B"), ] ) config = RegressionCheckConfig( @@ -344,13 +347,8 @@ def test_attribution_uses_the_true_immediately_preceding_run_across_a_version_ch tmp_path, ) -> None: """A one-off run on a different version (v5) sits between a v4 run and - the latest v4 run. `methodology_fingerprint` (version-inclusive) - excludes that v5 run from `baseline_runs`, so picking attribution's - comparison partner from `baseline_runs` (the old behavior) would - silently skip over it and compare against the older v4 run instead - - hiding the fact that something changed in between. Attribution must - use the coarser `lineage_key` instead, so it diffs against the run - that's actually immediately before `latest`. + the latest v4 run. `previous_run` must be v5 - the run actually + immediately before `latest` - not the older v4 run two steps back. """ v4_path = tmp_path / "v4" / "results.json" v5_path = tmp_path / "v5" / "results.json" @@ -393,8 +391,6 @@ def test_attribution_uses_the_true_immediately_preceding_run_across_a_version_ch ), ] ) - # min_runs=2 so the fingerprint-gated `baseline_runs` (just the v4 run) - # already satisfies `len(baseline_runs) + 1 >= min_runs` on its own. config = RegressionCheckConfig(metrics=["accuracy"], threshold_drop=0.10, min_runs=2) findings = run_regression_check(history, config) @@ -406,3 +402,67 @@ def test_attribution_uses_the_true_immediately_preceding_run_across_a_version_ch assert insight["from_commit"]["sha"] == "2" * 40 fields = {c["field"] for c in insight["changed_inputs"]} assert "system_prompt" in fields + + +def test_first_run_after_a_version_bump_can_still_be_detected_and_attributed( + tmp_path, +) -> None: + """The canonical scenario this whole feature targets: a run regresses + right after a version bump, on the very first run of the new version - + there is no prior run yet sharing that new version's + `methodology_fingerprint`. Gating `baseline_runs` on the fingerprint + (version-inclusive) would make this check structurally unable to fire + at all here (no baseline shares the brand-new fingerprint); gating on + the coarser, version-blind `lineage_key` lets the two older v3 runs + serve as the baseline for the fresh v4 regression, exactly as + research.md's own "run v3 -> v4" example describes. + """ + v3_a_path = tmp_path / "v3a" / "results.json" + v3_b_path = tmp_path / "v3b" / "results.json" + v4_path = tmp_path / "v4" / "results.json" + _write_result_json( + v3_a_path, accuracy=0.90, version="3", deployment="gpt-4o", commit_sha="1" * 40 + ) + _write_result_json( + v3_b_path, accuracy=0.91, version="3", deployment="gpt-4o", commit_sha="2" * 40 + ) + _write_result_json( + v4_path, accuracy=0.70, version="4", deployment="gpt-4o-mini", commit_sha="3" * 40 + ) + + history = ResultsHistory( + runs=[ + _run( + {"accuracy": 0.90}, + run_id="v3a", + offset_days=-2, + lineage_key="L", + raw_path=v3_a_path, + ), + _run( + {"accuracy": 0.91}, + run_id="v3b", + offset_days=-1, + lineage_key="L", + raw_path=v3_b_path, + ), + _run( + {"accuracy": 0.70}, + run_id="v4", + offset_days=0, + lineage_key="L", + raw_path=v4_path, + ), + ] + ) + config = RegressionCheckConfig(metrics=["accuracy"], threshold_drop=0.10, min_runs=3) + + findings = run_regression_check(history, config) + + assert len(findings) == 1 + insight = findings[0].evidence["insight"] + # Attributed to the v3 -> v4 change, not silently skipped. + assert insight["from_commit"]["sha"] == "2" * 40 + assert insight["to_commit"]["sha"] == "3" * 40 + fields = {c["field"] for c in insight["changed_inputs"]} + assert fields == {"system_prompt", "model"} diff --git a/tests/unit/test_agent_results_history.py b/tests/unit/test_agent_results_history.py index 23d7be3f..b18e014b 100644 --- a/tests/unit/test_agent_results_history.py +++ b/tests/unit/test_agent_results_history.py @@ -9,7 +9,7 @@ from types import SimpleNamespace from agentops.agent.config import FoundryControlSourceConfig, ResultsHistorySourceConfig -from agentops.agent.sources.results_history import collect_results_history +from agentops.agent.sources.results_history import _lineage_key, collect_results_history def _write_run( @@ -37,6 +37,33 @@ def _write_run( (run_dir / "results.json").write_text(json.dumps(payload), encoding="utf-8") +def test_lineage_key_ignores_deployment_for_direct_model_targets() -> None: + """A `model:` target has no `name`/`url` - only `raw` + (which embeds the deployment itself, e.g. `"model:gpt-4o"`) and + `deployment`. Falling back to `raw` for the lineage identity would + make a deployment change look like a different agent entirely, + defeating the point of a version/deployment-blind key - needed so + `agent.checks.regression` can pick the true immediately-preceding run + across a model-deployment change, same as a Foundry prompt version + bump.""" + gpt4o = { + "target": {"kind": "model_direct", "raw": "model:gpt-4o", "deployment": "gpt-4o"}, + "dataset_path": "data/smoke.jsonl", + "evaluators": ["CoherenceEvaluator"], + } + gpt4o_mini = { + "target": { + "kind": "model_direct", + "raw": "model:gpt-4o-mini", + "deployment": "gpt-4o-mini", + }, + "dataset_path": "data/smoke.jsonl", + "evaluators": ["CoherenceEvaluator"], + } + + assert _lineage_key(gpt4o) == _lineage_key(gpt4o_mini) + + def test_collect_results_history_orders_oldest_to_newest(tmp_path: Path) -> None: workspace = tmp_path results = workspace / ".agentops" / "results" diff --git a/tests/unit/test_cockpit.py b/tests/unit/test_cockpit.py index 503783dd..52c9feab 100644 --- a/tests/unit/test_cockpit.py +++ b/tests/unit/test_cockpit.py @@ -1934,6 +1934,92 @@ def test_version_history_names_changes_vs_previous_run(tmp_path: Path): assert second["regressed"] is True +def _write_direct_model_eval_run( + workspace: Path, + *, + timestamp_dir: str, + deployment: str, + accuracy: float, + commit_sha: str, + started_at: str, +) -> None: + """A ``model:`` target has no ``name``/``url`` - only + ``raw`` (which embeds the deployment itself, e.g. ``"model:gpt-4o"``) + and ``deployment``.""" + out = workspace / ".agentops" / "results" / timestamp_dir + out.mkdir(parents=True, exist_ok=True) + payload = { + "version": 1, + "started_at": started_at, + "finished_at": started_at, + "duration_seconds": 1.0, + "target": { + "kind": "model_direct", + "raw": f"model:{deployment}", + "deployment": deployment, + }, + "dataset_path": "data/smoke.jsonl", + "evaluators": ["CoherenceEvaluator"], + "rows": [], + "aggregate_metrics": {"accuracy": accuracy}, + "thresholds": [], + "summary": { + "items_total": 1, + "items_passed_all": 1, + "items_pass_rate": 1.0, + "thresholds_total": 0, + "thresholds_passed": 0, + "threshold_pass_rate": 1.0, + "overall_passed": True, + }, + "config": {}, + "commit": { + "sha": commit_sha, + "short_sha": commit_sha[:7], + "subject": "A commit", + "author": "Dev", + "authored_at": started_at, + "source": "ci", + }, + } + (out / "results.json").write_text(json.dumps(payload), encoding="utf-8") + + +def test_version_lineage_key_ignores_deployment_for_direct_model_targets( + tmp_path: Path, +): + """``raw`` embeds the deployment for `model:` targets + (unlike `name`/`url` for every other target kind) - falling back to it + for the version-lineage identity would make a deployment change start + a brand new lineage instead of being detected as a change within the + same one, defeating the entire point of this view.""" + _write_direct_model_eval_run( + tmp_path, + timestamp_dir="2026-09-01T10-00-00Z", + deployment="gpt-4o", + accuracy=0.91, + commit_sha="a" * 40, + started_at="2026-09-01T10:00:00+00:00", + ) + _write_direct_model_eval_run( + tmp_path, + timestamp_dir="2026-09-10T14-03-00Z", + deployment="gpt-4o-mini", + accuracy=0.79, + commit_sha="b" * 40, + started_at="2026-09-10T14:03:00+00:00", + ) + + runs = _load_eval_runs(tmp_path) + + assert len(runs) == 2 + first, second = runs + assert first["version_lineage_key"] == second["version_lineage_key"] + fields = {c["field"] for c in second["changed_inputs"]} + assert fields == {"model"} + assert second["regressed"] is True + + def test_version_history_latency_improvement_is_not_flagged_as_regressed( tmp_path: Path, ): @@ -2098,8 +2184,9 @@ def test_project_run_is_cached_by_path_and_mtime(tmp_path: Path, monkeypatch): """Re-rendering the cockpit without any new run must not re-parse and re-validate (``RunResult.model_validate``) the same unchanged ``results.json`` files again - that cost is paid once per file, not - once per render.""" - cockpit_module._PROJECT_RUN_CACHE.clear() + once per render, as long as the caller passes the same explicit + ``run_cache`` dict both times (there is no implicit/global cache).""" + run_cache: dict = {} _write_full_eval_run( tmp_path, timestamp_dir="2026-09-01T10-00-00Z", @@ -2117,16 +2204,42 @@ def _counting_uncached(path, *, run_id): monkeypatch.setattr(cockpit_module, "_project_run_uncached", _counting_uncached) - first = _load_eval_runs(tmp_path) - second = _load_eval_runs(tmp_path) + first = _load_eval_runs(tmp_path, run_cache=run_cache) + second = _load_eval_runs(tmp_path, run_cache=run_cache) assert calls["count"] == 1 assert first[0]["run_id"] == second[0]["run_id"] assert first[0]["commit"] == second[0]["commit"] +def test_project_run_without_cache_reparses_every_call(tmp_path: Path, monkeypatch): + """The default (``run_cache=None``) does no caching at all - a caller + that never opts in never gets stale data, by construction.""" + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-01T10-00-00Z", + accuracy=0.91, + commit_sha="a" * 40, + started_at="2026-09-01T10:00:00+00:00", + ) + + calls = {"count": 0} + original = cockpit_module._project_run_uncached + + def _counting_uncached(path, *, run_id): + calls["count"] += 1 + return original(path, run_id=run_id) + + monkeypatch.setattr(cockpit_module, "_project_run_uncached", _counting_uncached) + + _load_eval_runs(tmp_path) + _load_eval_runs(tmp_path) + + assert calls["count"] == 2 + + def test_project_run_cache_invalidated_on_file_change(tmp_path: Path, monkeypatch): - cockpit_module._PROJECT_RUN_CACHE.clear() + run_cache: dict = {} _write_full_eval_run( tmp_path, timestamp_dir="2026-09-01T10-00-00Z", @@ -2135,7 +2248,7 @@ def test_project_run_cache_invalidated_on_file_change(tmp_path: Path, monkeypatc started_at="2026-09-01T10:00:00+00:00", ) - first = _load_eval_runs(tmp_path) + first = _load_eval_runs(tmp_path, run_cache=run_cache) assert first[0]["metrics"]["accuracy"] == 0.91 results_path = tmp_path / ".agentops" / "results" / "2026-09-01T10-00-00Z" / "results.json" @@ -2147,7 +2260,7 @@ def test_project_run_cache_invalidated_on_file_change(tmp_path: Path, monkeypatc new_mtime = results_path.stat().st_mtime + 1 os.utime(results_path, (new_mtime, new_mtime)) - second = _load_eval_runs(tmp_path) + second = _load_eval_runs(tmp_path, run_cache=run_cache) assert second[0]["metrics"]["accuracy"] == 0.42 From 58838bb7c654d6b9e6cbf615e484bc038c1a0240 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 10:16:41 -0300 Subject: [PATCH 19/30] fix(pipeline): fall back to GITHUB_SHA when PR head isn't resolvable; --- src/agentops/pipeline/commit_info.py | 16 ++++- src/agentops/pipeline/comparison.py | 4 ++ src/agentops/pipeline/orchestrator.py | 30 +++++++++- tests/unit/test_commit_info.py | 30 ++++++++++ tests/unit/test_pipeline_publisher.py | 84 ++++++++++++++++++++++++++- 5 files changed, 156 insertions(+), 8 deletions(-) diff --git a/src/agentops/pipeline/commit_info.py b/src/agentops/pipeline/commit_info.py index b0ab0f12..395b9d35 100644 --- a/src/agentops/pipeline/commit_info.py +++ b/src/agentops/pipeline/commit_info.py @@ -51,10 +51,20 @@ def _pull_request_head_sha() -> Optional[str]: return sha or None -def _ci_git_sha() -> Optional[str]: +def _ci_git_sha(*, workspace: Optional[Path] = None) -> Optional[str]: if os.environ.get("GITHUB_EVENT_NAME") == "pull_request": head_sha = _pull_request_head_sha() - if head_sha: + # The generated PR workflow's default `actions/checkout` has no + # `ref`/`fetch-depth` override, so it checks out only the + # synthetic merge commit GITHUB_SHA points at - the real PR head + # commit object usually isn't present locally at all. Using + # head_sha anyway would make the later `git show` fail and commit + # capture return nothing for the whole run, which is worse than + # today's (wrong but resolvable) GITHUB_SHA attribution. Prefer + # head_sha only when it's actually resolvable in this checkout + # (e.g. the user customized the workflow to fetch full history or + # the head ref); otherwise fall through to GITHUB_SHA below. + if head_sha and commit_exists_locally(head_sha, workspace=workspace): return head_sha for env_var in _CI_SHA_ENV_VARS: value = os.environ.get(env_var) @@ -92,7 +102,7 @@ def resolve_commit_info(*, workspace: Optional[Path] = None) -> Optional[CommitI ``None`` - never raises - when neither source yields a resolvable commit, e.g. outside a git repository or when ``git`` is unavailable. """ - sha = _ci_git_sha() + sha = _ci_git_sha(workspace=workspace) source: Literal["ci", "local"] = "ci" if not sha: sha = _run_git(["rev-parse", "HEAD"], cwd=workspace) diff --git a/src/agentops/pipeline/comparison.py b/src/agentops/pipeline/comparison.py index 11da57e8..7b06f51d 100644 --- a/src/agentops/pipeline/comparison.py +++ b/src/agentops/pipeline/comparison.py @@ -122,6 +122,10 @@ def build_comparison( # `current` is still mid-orchestration here (persisted # after this call, Classic Foundry `publish: true` later # still) - only its own in-memory config can be checked. + # For `execution: local` + `publish: true`, + # orchestrator._publish_to_foundry_safely patches this + # value in and re-persists once that later publish step + # actually completes, so it isn't permanently `None`. from_report_url=resolve_report_url(baseline, results_path=baseline_path), to_report_url=resolve_report_url(current), ) diff --git a/src/agentops/pipeline/orchestrator.py b/src/agentops/pipeline/orchestrator.py index cfafd736..ae5df102 100644 --- a/src/agentops/pipeline/orchestrator.py +++ b/src/agentops/pipeline/orchestrator.py @@ -277,7 +277,13 @@ def _run_evaluation_local_snapshot( # Local execution only ever publishes to Classic Foundry. Cloud # execution goes through _run_evaluation_cloud and never reaches here. if config.publish_target() == "foundry": - _publish_to_foundry_safely(result, config, options.output_dir, progress=progress) + _publish_to_foundry_safely( + result, + config, + options.output_dir, + workspace=options.config_path.parent, + progress=progress, + ) return result @@ -773,9 +779,17 @@ def _publish_to_foundry_safely( config: AgentOpsConfig, output_dir: Path, *, + workspace: Path, progress: Optional[Callable[[str], None]] = None, ) -> None: - """Best-effort Classic Foundry publish. Failures are logged, never fatal.""" + """Best-effort Classic Foundry publish. Failures are logged, never fatal. + + On success, also patches ``result.comparison.insight.to_report_url`` + (left ``None`` by ``_persist``, since this run's own report_url only + exists after this publish step completes) and re-persists, so this + run's own already-written results.json/report.md don't permanently + miss the link to its own Foundry Evaluations page. + """ if config.publish_target() != "foundry": return @@ -806,6 +820,18 @@ def _publish_to_foundry_safely( ), encoding="utf-8", ) + + # `_persist` already wrote results.json/report.md with + # comparison.insight.to_report_url=None, since this publish step (the + # only source of this run's own report_url for `publish: true`) runs + # after persistence. Patch it in now and re-persist, rather than + # leaving the already-published run's own link permanently missing + # from its own results.json/report.md. + insight = result.comparison.insight if result.comparison is not None else None + if insight is not None and insight.to_report_url is None: + insight.to_report_url = published.studio_url + _persist(result, output_dir, workspace=workspace, commit_resolution_attempted=True) + notify( f"Published to {style('Classic Foundry Evaluations', 'bold')}: " f"{style(published.studio_url, 'cyan')}" diff --git a/tests/unit/test_commit_info.py b/tests/unit/test_commit_info.py index 70eaee39..d232b163 100644 --- a/tests/unit/test_commit_info.py +++ b/tests/unit/test_commit_info.py @@ -198,6 +198,36 @@ def test_pull_request_event_falls_back_to_github_sha_when_payload_unusable( assert info.source == "ci" +def test_pull_request_event_falls_back_to_github_sha_when_head_not_resolvable( + tmp_path, monkeypatch +): + """The generated PR workflow's default checkout has no `ref`/ + `fetch-depth` override, so the real PR head commit object usually + isn't present locally - only the merge commit GITHUB_SHA points at + is. Using an unresolvable head_sha anyway would make `git show` fail + and commit capture return nothing for the whole run (worse than the + wrong-but-resolvable GITHUB_SHA attribution this replaces), so it must + fall back to GITHUB_SHA instead.""" + repo = _init_repo(tmp_path) + merge_commit_sha = _run(["rev-parse", "HEAD"], cwd=repo) + unresolvable_head_sha = "9" * 40 # not a real commit in this repo + + event_path = tmp_path / "event.json" + _write_pull_request_event(event_path, head_sha=unresolvable_head_sha) + + monkeypatch.setenv("GITHUB_EVENT_NAME", "pull_request") + monkeypatch.setenv("GITHUB_EVENT_PATH", str(event_path)) + monkeypatch.setenv("GITHUB_SHA", merge_commit_sha) + monkeypatch.delenv("BUILD_SOURCEVERSION", raising=False) + monkeypatch.delenv("Build.SourceVersion", raising=False) + + info = commit_info.resolve_commit_info(workspace=repo) + + assert info is not None + assert info.sha == merge_commit_sha + assert info.source == "ci" + + def test_non_pull_request_event_still_uses_github_sha_directly( tmp_path, monkeypatch ): diff --git a/tests/unit/test_pipeline_publisher.py b/tests/unit/test_pipeline_publisher.py index 0d32527e..26242f23 100644 --- a/tests/unit/test_pipeline_publisher.py +++ b/tests/unit/test_pipeline_publisher.py @@ -2,6 +2,7 @@ from __future__ import annotations +import json import sys import types from pathlib import Path @@ -11,6 +12,10 @@ import pytest from agentops.core.results import ( + CommitInfo, + ComparisonInfo, + RegressedMetric, + RegressionInsight, RowMetric, RowResult, RunResult, @@ -165,7 +170,7 @@ def test_orchestrator_skips_publish_when_disabled(tmp_path: Path): result = _build_run_result() with mock.patch.object(publisher, "publish_to_foundry") as fake: - orchestrator._publish_to_foundry_safely(result, config, output_dir) + orchestrator._publish_to_foundry_safely(result, config, output_dir, workspace=tmp_path) fake.assert_not_called() # never reached because publish is None # The helper itself only runs when publish == "foundry"; we verify the @@ -190,7 +195,7 @@ def test_orchestrator_swallows_publish_errors(tmp_path: Path): publisher, "publish_to_foundry", side_effect=ImportError("no SDK") ): # Must not raise. - orchestrator._publish_to_foundry_safely(result, config, output_dir) + orchestrator._publish_to_foundry_safely(result, config, output_dir, workspace=tmp_path) assert not (output_dir / "cloud_evaluation.json").exists() @@ -215,7 +220,7 @@ def test_orchestrator_writes_cloud_evaluation_metadata(tmp_path: Path): evaluation_name="agentops-eval-abc", ) with mock.patch.object(publisher, "publish_to_foundry", return_value=fake_publish): - orchestrator._publish_to_foundry_safely(result, config, output_dir) + orchestrator._publish_to_foundry_safely(result, config, output_dir, workspace=tmp_path) meta_path = output_dir / "cloud_evaluation.json" assert meta_path.exists() @@ -225,6 +230,79 @@ def test_orchestrator_writes_cloud_evaluation_metadata(tmp_path: Path): assert payload["evaluation_name"] == "agentops-eval-abc" +def _commit(sha: str) -> CommitInfo: + return CommitInfo( + sha=sha, + short_sha=sha[:7], + subject="A commit", + author="Dev", + authored_at="2026-09-01T10:00:00+00:00", + source="ci", + ) + + +def test_orchestrator_patches_to_report_url_after_publish_completes(tmp_path: Path): + """``build_comparison`` can only leave ``to_report_url=None`` for the + run it's building - that run's own publish (if any) hasn't happened + yet. Once this publish step completes, the already-persisted + results.json/report.md must be patched with the real link rather than + permanently missing it.""" + from agentops.core.agentops_config import AgentOpsConfig + from agentops.pipeline import orchestrator, reporter + + config = AgentOpsConfig( + version=1, + agent="model:gpt-4o-mini", + dataset=Path("dataset.jsonl"), + publish=True, + project_endpoint="https://contoso.services.ai.azure.com/api/projects/p", + ) + output_dir = tmp_path / "out" + output_dir.mkdir() + result = _build_run_result() + result.comparison = ComparisonInfo( + baseline_path=".agentops/baseline/results.json", + metrics=[], + insight=RegressionInsight( + from_run_id="2026-09-01T10:00:00+00:00", + to_run_id="2026-09-10T14:03:00+00:00", + from_commit=_commit("a" * 40), + to_commit=_commit("b" * 40), + regressed_metrics=[ + RegressedMetric(metric="f1_score", from_value=0.91, to_value=0.75) + ], + changed_inputs=[], + explanation="Run aaaaaaa -> bbbbbbb: f1_score dropped from 0.91 to 0.75.", + commits_available_locally=False, + from_report_url="https://ai.azure.com/foundry/baseline", + to_report_url=None, + ), + ) + # Simulate _persist already having run before this publish step, with + # to_report_url still None at that point. + (output_dir / "results.json").write_text( + json.dumps(result.model_dump(mode="json"), indent=2), encoding="utf-8" + ) + (output_dir / "report.md").write_text(reporter.render(result), encoding="utf-8") + + fake_publish = publisher.PublishResult( + studio_url="https://ai.azure.com/projects/p/evaluations/current", + evaluation_name="agentops-eval-current", + ) + with mock.patch.object(publisher, "publish_to_foundry", return_value=fake_publish): + orchestrator._publish_to_foundry_safely(result, config, output_dir, workspace=tmp_path) + + assert result.comparison.insight.to_report_url == ( + "https://ai.azure.com/projects/p/evaluations/current" + ) + persisted = json.loads((output_dir / "results.json").read_text(encoding="utf-8")) + assert persisted["comparison"]["insight"]["to_report_url"] == ( + "https://ai.azure.com/projects/p/evaluations/current" + ) + report_text = (output_dir / "report.md").read_text(encoding="utf-8") + assert "https://ai.azure.com/projects/p/evaluations/current" in report_text + + def test_run_evaluation_cloud_uses_cloud_runner_and_does_not_invoke_locally( tmp_path: Path, ): From a43616485b680f7f454679cd1541e66652b4c282 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 10:16:58 -0300 Subject: [PATCH 20/30] docs: fix Cockpit eval-history payload contract to match implementation --- .../contracts/report-and-cockpit.md | 107 +++++++++++++----- 1 file changed, 80 insertions(+), 27 deletions(-) diff --git a/specs/012-regression-commit-attribution/contracts/report-and-cockpit.md b/specs/012-regression-commit-attribution/contracts/report-and-cockpit.md index ffb82e85..9c83e3f6 100644 --- a/specs/012-regression-commit-attribution/contracts/report-and-cockpit.md +++ b/specs/012-regression-commit-attribution/contracts/report-and-cockpit.md @@ -12,15 +12,39 @@ vs Baseline" section (`_render_comparison`), and only when ## Regression Insight -Run v3 → v4: **accuracy** dropped from 0.91 to 0.79. +**Regressed metrics:** +- `similarity`: 4.000 → 3.000 +- `accuracy`: 0.910 → 0.790 -**Likely cause:** the system prompt changed and the model changed from -`gpt-4o` to `gpt-4o-mini`. +Run v3 → v4: similarity dropped from 4.00 to 3.00 and accuracy dropped from +0.91 to 0.79. Likely cause: the system prompt changed and the model changed +from `gpt-4o` to `gpt-4o-mini`. **Suggested action:** review the prompt change and the model swap; consider reverting one at a time to isolate the cause. + +- [Baseline run in Foundry](https://ai.azure.com/...) +- [Regressed run in Foundry](https://ai.azure.com/...) ``` +Every metric that regressed is listed - not just the single worst one - in +`RegressionInsight.regressed_metrics`, ordered worst-first +(direction-aware, so a lower-is-better metric like `avg_latency_seconds` +getting worse still sorts as a regression). The "Regressed metrics" bullet +list only renders when there's more than one; a single-metric regression +keeps the original one-line prose shape. The two Foundry links +(`from_report_url`/`to_report_url`) render only when that side was +published (`execution: cloud`, or a completed local `publish: true`) - +omitted silently otherwise, never fabricated. For the live `--baseline` +comparison specifically, the *current* run's own link is initially `None` +when the run will go on to `publish: true`, because that publish step +happens after this comparison is built and persisted - but +`orchestrator._publish_to_foundry_safely` patches the value in and +re-persists `results.json`/`report.md` once that publish step actually +completes, so the already-written files don't permanently miss their own +link. Doctor's rolling check, which compares two already-completed +historical runs, never has this gap in the first place. + This section is purely additive to the report: it never replaces the existing Metrics/Thresholds/Comparison/Rows sections, and its absence (no regression, or missing commit metadata on either side) leaves `report.md` @@ -34,14 +58,21 @@ same explanation text via `Finding.evidence["insight"]`, and Doctor's own Markdown rendering already surfaces `Finding.summary`/`recommendation` — this plan extends that finding's `recommendation` text with the same deterministic explanation when an insight is available, in place of today's -generic "inspect prompt/model/dataset changes" instruction. +generic "inspect prompt/model/dataset changes" instruction. When no insight +could be built at all (e.g. Doctor fell back to a Foundry cloud-fetched run +with no local `results.json`, so no commit), the recommendation instead +names why attribution is unavailable +(`Finding.evidence["attribution_unavailable"]`) rather than silently +omitting it. ## Cockpit: version history No new HTTP route. The existing partial-load endpoint -(`GET /?_partial=1`, `cockpit.py:5405`) response gains a new section -alongside the existing eval "cards" built by `_build_eval_section` -(`cockpit.py:180`): +(`GET /?_partial=1`) response gains a new `eval_history` section +(`_build_eval_history_section`), alongside the existing eval "cards" +section. This is the actual payload shape (`cockpit.py`'s +`_build_eval_history_section`/`_attach_version_history`), **newest run +first**: ```jsonc { @@ -49,35 +80,57 @@ alongside the existing eval "cards" built by `_build_eval_section` "eval_history": { "has_runs": true, "entries": [ - { - "run_id": "20260901-101500", - "timestamp": "2026-09-01T10:15:00Z", - "commit": { "short_sha": "a1b2c3d", "subject": "..." }, - "metrics": { "accuracy": 0.91 }, - "methodology_fingerprint": "9f2a...", - "changed_inputs": [], - "regressed": false - }, { "run_id": "20260910-140300", "timestamp": "2026-09-10T14:03:00Z", - "commit": { "short_sha": "b7e91aa", "subject": "..." }, + "target": "greeter:4", "metrics": { "accuracy": 0.79 }, - "methodology_fingerprint": "9f2a...", + "commit_short_sha": "b7e91aa", + "commit_subject": "Swap eval agent to gpt-4o-mini", "changed_inputs": [ - { "field": "model", "description": "model changed from gpt-4o to gpt-4o-mini" } + { "field": "model", "description": "model changed from gpt-4o to gpt-4o-mini", "from_value": "gpt-4o", "to_value": "gpt-4o-mini" } ], - "regressed": true + "regressed": true, + "regressed_metrics": ["accuracy"], + "report_link": "/api/runs/20260910-140300/report", + "cloud_report_url": null, + "previous_cloud_report_url": "https://ai.azure.com/foundry/baseline" + }, + { + "run_id": "20260901-101500", + "timestamp": "2026-09-01T10:15:00Z", + "target": "greeter:3", + "metrics": { "accuracy": 0.91 }, + "commit_short_sha": "a1b2c3d", + "commit_subject": "Initial smoke eval", + "changed_inputs": [], + "regressed": false, + "regressed_metrics": [], + "report_link": "https://ai.azure.com/foundry/baseline", + "cloud_report_url": "https://ai.azure.com/foundry/baseline", + "previous_cloud_report_url": null } ] } } ``` -Entries are grouped/ordered by `methodology_fingerprint` (same grouping key -already used by `results_history.py` and Doctor's regression check), oldest -first. A run with no resolvable `commit` still appears, with `commit: null` -and an empty `changed_inputs` list rather than being omitted (per User Story -2's acceptance scenario 3). This view is read-only and introduces no writes -to any monitored resource, consistent with the constitution's Cockpit -read-only requirement. +Entries are grouped/compared using `version_lineage_key` - a coarser, +version/deployment-blind key (same agent identity, dataset, evaluators) +than Doctor's `methodology_fingerprint`, deliberately: this view exists to +show what changed *across* version bumps, where Doctor's rolling check +deliberately excludes a version bump from its own baseline. `commit` is +flattened to `commit_short_sha`/`commit_subject` (not a nested object); a +run with no resolvable commit still appears, with both `null` and an +empty `changed_inputs` list rather than being omitted (per User Story 2's +acceptance scenario 3). `regressed_metrics` names every metric that +regressed relative to the previous entry in its lineage, not just a +boolean - a lower-is-better metric (e.g. `avg_latency_seconds`) getting +*better* is never counted as a regression. `cloud_report_url` (this run's +own Foundry Evaluations link, `null` if never published) and +`previous_cloud_report_url` (the previous lineage-comparable run's, same +rule) let a regressed row link to both sides of the comparison; `report_link` +is unrelated pre-existing per-row routing (Foundry when published, else +this run's own local report page) and always non-null. This view is +read-only and introduces no writes to any monitored resource, consistent +with the constitution's Cockpit read-only requirement. From 2f3313f7c80ec5e7e9830ab458eb6430d384d70f Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 10:29:34 -0300 Subject: [PATCH 21/30] fix(cockpit): don't crash the whole page on a non-numeric metric value --- src/agentops/agent/cockpit.py | 15 +++++++++++---- tests/unit/test_cockpit.py | 25 +++++++++++++++++++++++++ 2 files changed, 36 insertions(+), 4 deletions(-) diff --git a/src/agentops/agent/cockpit.py b/src/agentops/agent/cockpit.py index d95b4f10..5c3efa24 100644 --- a/src/agentops/agent/cockpit.py +++ b/src/agentops/agent/cockpit.py @@ -4433,10 +4433,17 @@ def _render_eval_history_section(eval_history: Dict[str, Any]) -> str: commit_html = 'unknown commit' metrics = entry.get("metrics") or {} - metrics_html = ", ".join( - f"{_html_escape(str(name))}={value:.3f}" - for name, value in sorted(metrics.items()) - ) or "—" + # `_project_run_uncached` keeps projecting a run even when full + # `RunResult` validation fails, so a malformed/historical + # `results.json` can carry a non-numeric metric value here - + # formatting it with `:.3f` directly would raise and take down + # the whole Cockpit page for every entry, not just this one. + metric_parts: List[str] = [] + for name, value in sorted(metrics.items()): + numeric_value = _safe_float(value) + if numeric_value is not None: + metric_parts.append(f"{_html_escape(str(name))}={numeric_value:.3f}") + metrics_html = ", ".join(metric_parts) or "—" changes = entry.get("changed_inputs") or [] if changes: diff --git a/tests/unit/test_cockpit.py b/tests/unit/test_cockpit.py index 52c9feab..c2bdebd3 100644 --- a/tests/unit/test_cockpit.py +++ b/tests/unit/test_cockpit.py @@ -2294,6 +2294,31 @@ def test_cockpit_html_renders_version_history_section(tmp_path: Path): assert "regressed" in html +def test_version_history_tolerates_non_numeric_metric_value(tmp_path: Path): + """`_project_run_uncached` keeps projecting a run even when full + `RunResult` validation fails, so a malformed/historical results.json + can carry a non-numeric metric value - rendering must skip it rather + than raise and take down the whole page.""" + _write_full_eval_run( + tmp_path, + timestamp_dir="2026-09-01T10-00-00Z", + accuracy=0.91, + commit_sha="a" * 40, + started_at="2026-09-01T10:00:00+00:00", + ) + results_path = tmp_path / ".agentops" / "results" / "2026-09-01T10-00-00Z" / "results.json" + payload = json.loads(results_path.read_text(encoding="utf-8")) + payload["aggregate_metrics"]["coherence"] = "n/a" + results_path.write_text(json.dumps(payload), encoding="utf-8") + + cockpit_payload = build_cockpit_payload(tmp_path) + html = render_cockpit_html(cockpit_payload) # must not raise + + assert "Evaluation Version History" in html + assert "accuracy=0.910" in html + assert "coherence=" not in html + + def test_cockpit_html_version_history_empty_state(tmp_path: Path): payload = build_cockpit_payload(tmp_path) html = render_cockpit_html(payload) From b7f50842df2b56d8f7b53e5c3039493faa5211c8 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 10:29:52 -0300 Subject: [PATCH 22/30] docs: align spec/research/data-model with the shipped version-blind lineage key --- .../data-model.md | 12 ++-- .../research.md | 55 ++++++++++++++----- .../012-regression-commit-attribution/spec.md | 8 +-- 3 files changed, 51 insertions(+), 24 deletions(-) diff --git a/specs/012-regression-commit-attribution/data-model.md b/specs/012-regression-commit-attribution/data-model.md index 9aecb663..9d745fac 100644 --- a/specs/012-regression-commit-attribution/data-model.md +++ b/specs/012-regression-commit-attribution/data-model.md @@ -76,18 +76,20 @@ untouched. ## Cockpit projection (not a persisted model — computed per request) -**VersionHistoryEntry** (dict shape returned by `_project_run()` / -consumed by the cockpit UI): +**VersionHistoryEntry** (dict shape returned by `_build_eval_history_section()` +/ consumed by the cockpit UI; see `contracts/report-and-cockpit.md` for the +full example payload): | Key | Type | Notes | |---|---|---| | `run_id` | `str` | Existing field. | | `timestamp` | `Optional[str]` | Existing field. | -| `commit` | `Optional[dict]` | New: `CommitInfo.model_dump()` when known. | +| `commit_short_sha` / `commit_subject` | `Optional[str]` | New: flattened from `CommitInfo` when known (not a nested `commit` object). | | `metrics` | `Dict[str, float]` | Existing field. | -| `methodology_fingerprint` | `Optional[str]` | New: reused from `results_history._methodology_fingerprint()` so entries can be grouped/ordered per lineage. | -| `changed_inputs` | `List[dict]` | New: `ChangedInput` list vs. the previous entry sharing the same fingerprint; empty for the first run of a fingerprint. | +| `changed_inputs` | `List[dict]` | New: `ChangedInput` list vs. the previous entry sharing the same `version_lineage_key` (**not** `methodology_fingerprint` - a deliberately coarser, version/deployment-blind key per FR-005/research.md #3, so a version bump is detected as a change within the same lineage rather than excluded from it); empty for the first run of a lineage. | | `regressed` | `bool` | New: whether any metric regressed vs. the previous entry (drives a visual marker; the list itself is not regression-gated per FR-012/User Story 2). | +| `regressed_metrics` | `List[str]` | New: names every metric that regressed vs. the previous entry, direction-aware (a lower-is-better metric improving is never counted) - not just the boolean. | +| `cloud_report_url` / `previous_cloud_report_url` | `Optional[str]` | New: this entry's and the previous lineage-comparable entry's Foundry Evaluations link, when published; `None` otherwise. | ## State / lifecycle notes diff --git a/specs/012-regression-commit-attribution/research.md b/specs/012-regression-commit-attribution/research.md index 3f4d1ebd..ed9f70f3 100644 --- a/specs/012-regression-commit-attribution/research.md +++ b/specs/012-regression-commit-attribution/research.md @@ -44,12 +44,33 @@ call-site-specific changes that could drift out of sync. ## 3. Which prior run to diff against for the causal explanation -**Decision**: Attribute a regression to the single most recent comparable run -(the immediately preceding entry sharing the same methodology fingerprint — -same agent target, dataset, evaluator set, per -`_methodology_fingerprint()` in `src/agentops/agent/sources/results_history.py:145`), -not the rolling mean that Doctor's existing regression check -(`src/agentops/agent/checks/regression.py`) uses for its drop-percentage math. +**Decision**: Attribute a regression to the single most recent comparable +run — the immediately preceding entry sharing the same *lineage*: same +dataset, evaluator set, and agent identity, but deliberately +version/deployment-*blind* (`_lineage_key()` in +`src/agentops/agent/sources/results_history.py`, and the equivalent +`_version_lineage_key()` in `src/agentops/agent/cockpit.py`) — not the +rolling mean that Doctor's existing regression check +(`src/agentops/agent/checks/regression.py`) uses for its drop-percentage +math, and not the stricter, version-inclusive `_methodology_fingerprint()` +either. + +Using the strict fingerprint here was the first implementation and is a +trap worth naming explicitly: it hashes the *whole* target including +version/deployment, so it treats a version bump as a different +methodology and excludes the pre-bump run from the comparable set - which +makes the product owner's own canonical example ("Corrida v3 → v4") +unreachable, since the v3 run would never be selected as the "most recent +comparable run" to diff v4 against (there would need to be a prior run +*already on v4* for the fingerprint-filtered set to be non-empty at all). +The two keys now coexist for different, non-overlapping purposes: the +fingerprint still gates the rolling-baseline mean (mixing versions there +would be genuinely spurious noise) and opex.py's flaky-metric check +(mixing versions there would inflate variance spuriously); the lineage +key is used everywhere a single "most recent comparable run" must be +picked for causal attribution or Cockpit's version-history grouping, +precisely because a version/deployment change is the attribution this +feature exists to surface, not a reason to exclude the pair. **Rationale**: The product owner's own example ("Corrida v3 → v4") compares adjacent versions. A rolling mean of several prior runs has no single @@ -121,16 +142,20 @@ rendered. This is the natural, lowest-effort integration point for User Story ## 7. Where Cockpit's history view plugs in **Decision**: Extend `_project_run()` -(`src/agentops/agent/cockpit.py:856`) to include the run's `commit` field (if +(`src/agentops/agent/cockpit.py`) to include the run's `commit` field (if present) and a computed `changed_inputs` list versus the previous entry -sharing the same methodology fingerprint, reusing the same field-diff helper -from #5. `_load_eval_runs()` (`cockpit.py:827`) already returns an ordered, -scanned list of runs from `.agentops/results/*/results.json` — the version -history view is a new rendering of that same list, not a new data source. - -**Rationale**: This is the same scan Doctor's regression check and -`results_history.py` already perform; no new persisted index or storage is -needed once every run carries its own `commit` field. +sharing the same *version-blind lineage key* (`_version_lineage_key()`, +not the stricter `methodology_fingerprint` - see decision #3 above for why +this view deliberately groups across version bumps instead of excluding +them), reusing the same field-diff helper from #5. `_load_eval_runs()` +already returns an ordered, scanned list of runs from +`.agentops/results/*/results.json` — the version history view is a new +rendering of that same list, not a new data source. + +**Rationale**: This is the same kind of scan Doctor's regression check and +`results_history.py` already perform (grouped by their own, separate +version-blind lineage key - see decision #3); no new persisted index or +storage is needed once every run carries its own `commit` field. **Alternatives considered**: A separate persisted history/index file — rejected as redundant once per-run files carry everything needed. diff --git a/specs/012-regression-commit-attribution/spec.md b/specs/012-regression-commit-attribution/spec.md index 0d45ee46..e86b1f1e 100644 --- a/specs/012-regression-commit-attribution/spec.md +++ b/specs/012-regression-commit-attribution/spec.md @@ -66,7 +66,7 @@ A developer running `agentops eval run` locally against a git-managed workspace - No prior comparable run exists yet for a given methodology (first run of its kind): no regression explanation and no history entry's "changed" comparison is produced for that first run. - A metric improves rather than regresses between two comparable runs: no cause/attribution explanation is generated for the report (attribution is regression-triggered), but the run still appears in Cockpit's version history with its changes relative to the previous run. - The regression is detected via the rolling-baseline check rather than an explicit `--baseline` comparison: the same attribution logic applies to either detection path. -- Two runs are "comparable" by methodology fingerprint but were produced through different execution modes (e.g. one local, one CI): attribution still proceeds as long as both have commit metadata. +- Two runs are "comparable" by the version-blind lineage rule (FR-005) but were produced through different execution modes (e.g. one local, one CI), or across a version/deployment change: attribution still proceeds as long as both have commit metadata. ## Requirements *(mandatory)* @@ -76,14 +76,14 @@ A developer running `agentops eval run` locally against a git-managed workspace - **FR-002**: For evaluation runs against a Foundry hosted agent executed via cloud or azd execution in a CI environment, the system MUST reliably capture commit metadata from CI-provided source-control information, at the same reliability level already achieved for existing CI-based commit capture. - **FR-003**: For local execution, the system MUST make a best-effort attempt to capture commit metadata from the local git repository when the working directory is one, and MUST allow the run to complete normally, without error, when it is not. - **FR-004**: Commit metadata MUST be persisted as an additive part of the run's stored result such that existing consumers of that result format that are unaware of the new field are unaffected. -- **FR-005**: The system MUST only compare two runs for regression-cause attribution when they share the same evaluation methodology (same agent target, dataset, and evaluator set) already used for existing baseline and regression comparisons. +- **FR-005**: The system MUST only compare two runs for regression-cause attribution when they share the same dataset and evaluator set, and the same agent identity — but deliberately *independent of* the agent's version/model/deployment, since a change in exactly those fields is the attribution this feature exists to surface. This is a coarser, version-blind comparability rule distinct from the stricter, version-inclusive methodology fingerprint already used for the rolling-baseline drop-percentage calculation and other existing checks (e.g. flaky-metric detection), where mixing versions would be genuinely spurious. - **FR-006**: When a regression is detected between two comparable runs that both have recorded commit metadata, the system MUST determine, at minimum, whether the evaluated agent's system prompt content, its model/deployment identifier, and other tracked run configuration (dataset, evaluator set, thresholds) changed between the two commits. - **FR-007**: When a regression cause is determined, the system MUST produce a plain-language, one-to-two-sentence explanation naming the regressed metric's before/after values and the concrete inputs found to have changed. - **FR-008**: Where a likely cause is identified, the system SHOULD include a brief suggested corrective adjustment tied directly to that cause. - **FR-009**: When commit metadata is missing for either compared run, or the changed inputs cannot be determined, the system MUST still report the regression's metric numbers as it does today, without producing a fabricated cause. - **FR-010**: A produced regression explanation MUST appear in the same evaluation report already generated for the run, and MUST reach every destination that report is already delivered to (CI artifact upload and PR comment) without requiring a new delivery mechanism. - **FR-011**: Adding commit metadata and regression explanations MUST NOT change the existing exit-code or threshold-gating outcome of a run; this feature is informational only. -- **FR-012**: Cockpit MUST provide a history view that, for a given evaluation methodology, lists its evaluated runs in order, each showing its commit reference (when known) and the changes detected relative to the immediately preceding comparable run, independent of whether that run regressed. +- **FR-012**: Cockpit MUST provide a history view that, for a given version lineage (same dataset, evaluator set, and agent identity, version/deployment-blind per FR-005), lists its evaluated runs in order, each showing its commit reference (when known) and the changes detected relative to the immediately preceding comparable run, independent of whether that run regressed. - **FR-013**: Determining what changed between two runs' commits MUST be based on deterministic comparison of recorded fields and content (prompt text, model identifier, tracked configuration values), not a generative or LLM-based summarization call. ### Key Entities @@ -106,7 +106,7 @@ A developer running `agentops eval run` locally against a git-managed workspace - "Hosted agents" refers to Foundry hosted agent targets evaluated via cloud or azd execution, consistent with how the system already classifies agent targets. - The CI environment used for the PR pipeline provides the commit SHA of the code under evaluation through standard CI environment variables, the same class of information already relied on for existing CI-based commit capture elsewhere in the system. -- "Comparable runs" reuses the existing methodology-fingerprint concept (same agent target, dataset, and evaluator set) already used for baseline and rolling-baseline regression comparisons; this feature does not introduce a new comparability rule. +- "Comparable runs" for regression-cause attribution (FR-005) and Cockpit's history view (FR-012) means a coarser, version/deployment-blind variant of the existing methodology-fingerprint concept: same dataset, evaluator set, and agent identity, but *not* requiring the same prompt version/model/deployment - the opposite of the stricter, version-inclusive fingerprint already used for the rolling-baseline drop-percentage math and other existing checks. This feature does introduce this new, coarser comparability rule specifically because the stricter one would exclude the very version/deployment change this feature exists to attribute. - A full commit-to-commit diff assumes both compared commits are reachable in the local git history available at comparison time; when they are not (e.g. a shallow clone), the system falls back to comparing the fields already recorded in each run's stored result instead of requiring direct git access to both commits. - "System prompt" content for diffing is whatever prompt content the run's configuration already records for the evaluated agent version; this feature does not introduce new prompt-capture beyond what configuration/version resolution already provides. - Suggested corrective adjustments are short and rule-based, directly tied to the specific change detected (e.g. a changed prompt suggests reviewing that prompt change); this is not a general-purpose troubleshooting assistant and does not call out to a language model. From 099557b5a65b26f34ca90be202fbf1a0be99defb Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 10:54:24 -0300 Subject: [PATCH 23/30] fix(doctor): keep rolling-baseline detection on methodology_fingerprint; --- src/agentops/agent/checks/regression.py | 53 +++++++----- tests/unit/test_agent_checks_regression.py | 96 +++++++--------------- 2 files changed, 64 insertions(+), 85 deletions(-) diff --git a/src/agentops/agent/checks/regression.py b/src/agentops/agent/checks/regression.py index 89cf0b14..422d7be7 100644 --- a/src/agentops/agent/checks/regression.py +++ b/src/agentops/agent/checks/regression.py @@ -37,7 +37,7 @@ def _load_run_result(summary: RunSummary) -> Optional[RunResult]: def _attribution_unavailable_reason( latest: RunSummary, - previous_run: RunSummary, + previous_run: Optional[RunSummary], latest_result: Optional[RunResult], previous_result: Optional[RunResult], ) -> str: @@ -51,6 +51,8 @@ def _attribution_unavailable_reason( ``agentops eval run`` (including ``execution: cloud``/``azd``), which always attempts commit capture. """ + if previous_run is None: + return "attribution unavailable: no comparable prior run found" non_local = [ run for run, result in ((latest, latest_result), (previous_run, previous_result)) @@ -79,34 +81,45 @@ def run_regression_check( return [] latest = runs[-1] - # Only compare against runs that share the same dataset, evaluators, - # and agent identity - but deliberately version/deployment-*blind* - # (`lineage_key`, not the finer `methodology_fingerprint` opex.py's - # flaky-metric check needs, which does want version included since - # mixing versions there would inflate variance spuriously). A version - # bump is exactly the kind of change this check - and its causal - # attribution - exists to catch, not an incompatible methodology to - # exclude: keying this on the fingerprint instead would make the - # check structurally unable to fire on the first run(s) after a - # version bump (there being no baseline yet sharing that new, - # not-yet-established fingerprint), making the feature's own - # canonical "run v3 -> v4" scenario unreachable. - lineage_key = latest.lineage_key - if lineage_key is None: + # Only compare against runs that share the same evaluation methodology + # (same agent target, dataset, and evaluator set) - this is Doctor's + # pre-existing rolling-baseline detection rule, unchanged by this + # feature. Changing *whether*/*how sensitively* a regression is + # detected is out of scope here: this feature only adds a causal + # explanation once a regression has already been detected the same + # way it always was. + fingerprint = latest.methodology_fingerprint + if fingerprint is None: baseline_runs = runs[:-1] else: - baseline_runs = [r for r in runs[:-1] if r.lineage_key == lineage_key] + baseline_runs = [ + r for r in runs[:-1] if r.methodology_fingerprint == fingerprint + ] if len(baseline_runs) + 1 < config.min_runs: return [] if not baseline_runs: return [] # The immediately preceding comparable run, for causal attribution - - # distinct from the rolling drop% mean below, which uses every - # comparable run, not just the last one. - previous_run = baseline_runs[-1] + # distinct from `baseline_runs` above (detection's rolling drop% mean, + # left untouched). Deliberately keyed on the coarser `lineage_key` + # (dataset, evaluators, agent identity - version/deployment excluded) + # rather than `baseline_runs`/`methodology_fingerprint`: when a + # regression *has* been detected and happens to coincide with a + # version bump, the fingerprint would exclude the pre-bump run from + # `baseline_runs`, so picking the comparison partner from there could + # silently skip over it and attribute the cause to an older, less + # relevant run instead. This only changes which run an *already-fired* + # finding is explained against, never whether one fires. + lineage_key = latest.lineage_key + if lineage_key is None: + lineage_runs = runs[:-1] + else: + lineage_runs = [r for r in runs[:-1] if r.lineage_key == lineage_key] + previous_run = lineage_runs[-1] if lineage_runs else None + latest_result = _load_run_result(latest) - previous_result = _load_run_result(previous_run) + previous_result = _load_run_result(previous_run) if previous_run is not None else None findings: List[Finding] = [] for metric in config.metrics: diff --git a/tests/unit/test_agent_checks_regression.py b/tests/unit/test_agent_checks_regression.py index 2b02b1c4..d23d9f76 100644 --- a/tests/unit/test_agent_checks_regression.py +++ b/tests/unit/test_agent_checks_regression.py @@ -125,18 +125,21 @@ def test_regression_check_skips_when_baseline_too_small() -> None: assert findings == [] -def test_regression_check_ignores_baselines_with_mismatched_lineage() -> None: - """Baselines from a different dataset/evaluator set (lineage) must not - count - unlike a version/deployment difference, which must (see +def test_regression_check_ignores_baselines_with_mismatched_methodology() -> None: + """Baselines from a different dataset/evaluator set must not count. + Detection (`baseline_runs`) stays on the strict, version-inclusive + `methodology_fingerprint` - unchanged by this feature; only picking + which run to *explain* an already-detected regression against uses the + coarser `lineage_key` (see ``test_attribution_uses_the_true_immediately_preceding_run_across_a_version_change`` below).""" history = ResultsHistory( runs=[ # These baselines used a different methodology (e.g. smoke dataset) # and must be excluded from the comparison. - _run({"coherence": 4.5}, run_id="b1", offset_days=-3, lineage_key="A"), - _run({"coherence": 4.5}, run_id="b2", offset_days=-2, lineage_key="A"), - _run({"coherence": 3.0}, run_id="latest", offset_days=0, lineage_key="B"), + _run({"coherence": 4.5}, run_id="b1", offset_days=-3, fingerprint="A"), + _run({"coherence": 4.5}, run_id="b2", offset_days=-2, fingerprint="A"), + _run({"coherence": 3.0}, run_id="latest", offset_days=0, fingerprint="B"), ] ) config = RegressionCheckConfig( @@ -146,14 +149,14 @@ def test_regression_check_ignores_baselines_with_mismatched_lineage() -> None: assert findings == [] -def test_regression_check_uses_matching_lineage_baselines() -> None: - """Baselines with the same lineage key as the latest run drive the check.""" +def test_regression_check_uses_matching_methodology_baselines() -> None: + """Baselines with the same fingerprint as the latest run drive the check.""" history = ResultsHistory( runs=[ - _run({"coherence": 4.5}, run_id="other", offset_days=-4, lineage_key="A"), - _run({"coherence": 4.5}, run_id="b1", offset_days=-3, lineage_key="B"), - _run({"coherence": 4.5}, run_id="b2", offset_days=-2, lineage_key="B"), - _run({"coherence": 3.0}, run_id="latest", offset_days=0, lineage_key="B"), + _run({"coherence": 4.5}, run_id="other", offset_days=-4, fingerprint="A"), + _run({"coherence": 4.5}, run_id="b1", offset_days=-3, fingerprint="B"), + _run({"coherence": 4.5}, run_id="b2", offset_days=-2, fingerprint="B"), + _run({"coherence": 3.0}, run_id="latest", offset_days=0, fingerprint="B"), ] ) config = RegressionCheckConfig( @@ -404,65 +407,28 @@ def test_attribution_uses_the_true_immediately_preceding_run_across_a_version_ch assert "system_prompt" in fields -def test_first_run_after_a_version_bump_can_still_be_detected_and_attributed( - tmp_path, -) -> None: - """The canonical scenario this whole feature targets: a run regresses - right after a version bump, on the very first run of the new version - - there is no prior run yet sharing that new version's - `methodology_fingerprint`. Gating `baseline_runs` on the fingerprint - (version-inclusive) would make this check structurally unable to fire - at all here (no baseline shares the brand-new fingerprint); gating on - the coarser, version-blind `lineage_key` lets the two older v3 runs - serve as the baseline for the fresh v4 regression, exactly as - research.md's own "run v3 -> v4" example describes. +def test_first_run_after_a_version_bump_is_a_known_accepted_gap_in_doctor() -> None: + """Doctor's rolling-baseline detection (`baseline_runs`/`methodology_fingerprint`) + is deliberately left untouched by this feature - changing *whether*/*how + sensitively* it fires is a bigger, separate decision than "add a causal + explanation to an already-detected regression", and would alter + detection behavior for every Doctor user, not just this feature's own + scenario. So the very first run after a version bump still can't fire + here: no prior run shares the brand-new fingerprint yet, so + `baseline_runs` is empty and the gate returns early - by design, not + oversight. The canonical "run v3 -> v4" scenario from research.md is + fully covered instead via the explicit `--baseline` comparison path + (`pipeline.comparison`), which has no such gate. """ - v3_a_path = tmp_path / "v3a" / "results.json" - v3_b_path = tmp_path / "v3b" / "results.json" - v4_path = tmp_path / "v4" / "results.json" - _write_result_json( - v3_a_path, accuracy=0.90, version="3", deployment="gpt-4o", commit_sha="1" * 40 - ) - _write_result_json( - v3_b_path, accuracy=0.91, version="3", deployment="gpt-4o", commit_sha="2" * 40 - ) - _write_result_json( - v4_path, accuracy=0.70, version="4", deployment="gpt-4o-mini", commit_sha="3" * 40 - ) - history = ResultsHistory( runs=[ - _run( - {"accuracy": 0.90}, - run_id="v3a", - offset_days=-2, - lineage_key="L", - raw_path=v3_a_path, - ), - _run( - {"accuracy": 0.91}, - run_id="v3b", - offset_days=-1, - lineage_key="L", - raw_path=v3_b_path, - ), - _run( - {"accuracy": 0.70}, - run_id="v4", - offset_days=0, - lineage_key="L", - raw_path=v4_path, - ), + _run({"accuracy": 0.90}, run_id="v3a", offset_days=-2, fingerprint="V3"), + _run({"accuracy": 0.91}, run_id="v3b", offset_days=-1, fingerprint="V3"), + _run({"accuracy": 0.70}, run_id="v4", offset_days=0, fingerprint="V4"), ] ) config = RegressionCheckConfig(metrics=["accuracy"], threshold_drop=0.10, min_runs=3) findings = run_regression_check(history, config) - assert len(findings) == 1 - insight = findings[0].evidence["insight"] - # Attributed to the v3 -> v4 change, not silently skipped. - assert insight["from_commit"]["sha"] == "2" * 40 - assert insight["to_commit"]["sha"] == "3" * 40 - fields = {c["field"] for c in insight["changed_inputs"]} - assert fields == {"system_prompt", "model"} + assert findings == [] From 6d10045ddf308f77c57dc0106a7ba5fa5ba5df42 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 10:54:44 -0300 Subject: [PATCH 24/30] fix(pipeline): capture commit before execution starts, not after it finishes --- src/agentops/pipeline/orchestrator.py | 45 +++++++++++-- .../test_regression_commit_attribution.py | 67 +++++++++++++++++++ 2 files changed, 106 insertions(+), 6 deletions(-) diff --git a/src/agentops/pipeline/orchestrator.py b/src/agentops/pipeline/orchestrator.py index ae5df102..d8d1acf2 100644 --- a/src/agentops/pipeline/orchestrator.py +++ b/src/agentops/pipeline/orchestrator.py @@ -32,6 +32,7 @@ select_evaluators, ) from agentops.core.results import ( + CommitInfo, RowMetric, RowResult, RunResult, @@ -150,6 +151,11 @@ def _run_evaluation_local_snapshot( started_at = datetime.now(timezone.utc) started_perf = time.perf_counter() + # Captured now, before the (potentially long-running) row-by-row loop + # below, not after it finishes - otherwise a commit or checkout made + # locally while this run is still executing would attribute the + # result to code that was never actually evaluated. + commit = resolve_commit_info(workspace=options.config_path.parent) target = classify_agent( options.agent_override or config.agent, @@ -265,7 +271,7 @@ def _run_evaluation_local_snapshot( }, ) - _finalize_commit_and_comparison(result, options) + _finalize_commit_and_comparison(result, options, commit=commit) _persist( result, @@ -322,6 +328,10 @@ def _run_evaluation_cloud_snapshot( """ started_at = datetime.now(timezone.utc) started_perf = time.perf_counter() + # Captured now, before Foundry's (potentially long-running) server-side + # run below, not after it completes - see the matching comment in + # _run_evaluation_local_snapshot. + commit = resolve_commit_info(workspace=options.config_path.parent) target = classify_agent( options.agent_override or config.agent, @@ -555,7 +565,7 @@ def _run_evaluation_cloud_snapshot( }, ) - _finalize_commit_and_comparison(result, options) + _finalize_commit_and_comparison(result, options, commit=commit) _persist( result, @@ -607,6 +617,10 @@ def _run_evaluation_azd( started_at = datetime.now(timezone.utc) progress = options.progress or (lambda _msg: None) workspace = options.config_path.parent + # Captured now, before azd's (potentially long-running) subprocess + # below, not after it finishes - see the matching comment in + # _run_evaluation_local_snapshot. + commit = resolve_commit_info(workspace=workspace) target = classify_agent( options.agent_override or config.agent, @@ -642,6 +656,7 @@ def _display(path: Path) -> str: resolution=resolution, workspace=workspace, started_at=started_at, + commit=commit, recipe_display=_display(resolution.path), ) return _run_evaluation_azd_legacy( @@ -650,6 +665,7 @@ def _display(path: Path) -> str: recipe_path=resolution.path, workspace=workspace, started_at=started_at, + commit=commit, recipe_display=_display(resolution.path), ) @@ -661,6 +677,7 @@ def _run_evaluation_azd_legacy( recipe_path: Path, workspace: Path, started_at: datetime, + commit: Optional[CommitInfo] = None, recipe_display: str, ) -> RunResult: """Legacy surface: ``azd ai agent eval`` via the original adapter.""" @@ -689,7 +706,7 @@ def _run_evaluation_azd_legacy( started_at=started_at, ) - _finalize_commit_and_comparison(result, options) + _finalize_commit_and_comparison(result, options, commit=commit) _persist( result, @@ -708,6 +725,7 @@ def _run_evaluation_azd_current( resolution: Any, workspace: Path, started_at: datetime, + commit: Optional[CommitInfo] = None, recipe_display: str, ) -> RunResult: """Current surface: ``azd ai eval`` via the current-surface adapter.""" @@ -762,7 +780,7 @@ def _run_evaluation_azd_current( resolution=resolution, ) - _finalize_commit_and_comparison(result, options) + _finalize_commit_and_comparison(result, options, commit=commit) _persist( result, @@ -1163,7 +1181,12 @@ def _summarize( # --------------------------------------------------------------------------- -def _finalize_commit_and_comparison(result: RunResult, options: RunOptions) -> None: +def _finalize_commit_and_comparison( + result: RunResult, + options: RunOptions, + *, + commit: Optional[CommitInfo] = None, +) -> None: """Attach commit metadata, then build the ``--baseline`` comparison. Commit metadata must be resolved before the comparison is built so a @@ -1171,10 +1194,20 @@ def _finalize_commit_and_comparison(result: RunResult, options: RunOptions) -> N ``pipeline.regression_insight.build_regression_insight``); resolving it only at persist time (after the comparison already ran) would always leave ``current.commit`` unset for this call. + + ``commit``, when given, is a value every current call site captures + *before* dispatching the (potentially long-running) evaluation/azd + subprocess/Foundry run - not resolving it here, after execution has + already finished, avoids attributing the result to whatever the repo's + HEAD happens to be by the time execution completes (which, for a local + run that takes minutes, is not guaranteed to still be the commit that + was actually evaluated). Falling back to resolving it now - the + previous behavior - stays as a safety net for a caller that doesn't + pre-capture it. """ workspace = options.config_path.parent if result.commit is None: - result.commit = resolve_commit_info(workspace=workspace) + result.commit = commit if commit is not None else resolve_commit_info(workspace=workspace) if options.baseline_path is not None: baseline = comparison_module.load_baseline(options.baseline_path) result.comparison = comparison_module.build_comparison( diff --git a/tests/integration/test_regression_commit_attribution.py b/tests/integration/test_regression_commit_attribution.py index 3a4c0e12..7ac9904c 100644 --- a/tests/integration/test_regression_commit_attribution.py +++ b/tests/integration/test_regression_commit_attribution.py @@ -11,6 +11,7 @@ from __future__ import annotations import json +import subprocess import threading from http.server import BaseHTTPRequestHandler, HTTPServer from pathlib import Path @@ -218,3 +219,69 @@ def test_regression_commit_attribution_purely_local( assert current_result.comparison.insight is not None report_text = (current_dir / "report.md").read_text(encoding="utf-8") assert "## Regression Insight" in report_text + + +def _run_git(args: list[str], *, cwd: Path) -> str: + completed = subprocess.run( + ["git", *args], cwd=cwd, capture_output=True, text=True, check=True + ) + return completed.stdout.strip() + + +def test_commit_is_captured_before_execution_not_after( + tmp_path: Path, monkeypatch, good_and_bad_servers +) -> None: + """A local run's commit must reflect the repo's HEAD when the run + *started*, not whatever HEAD happens to be once execution (the + row-by-row agent/evaluator loop, which can take a while) finishes - a + commit made to the repo mid-run must not retroactively change which + commit this run is attributed to. + """ + good_url, _bad_url = good_and_bad_servers + for env_var in ("GITHUB_SHA", "BUILD_SOURCEVERSION", "Build.SourceVersion"): + monkeypatch.delenv(env_var, raising=False) + + repo = tmp_path + _run_git(["init"], cwd=repo) + _run_git(["config", "user.email", "dev@example.com"], cwd=repo) + _run_git(["config", "user.name", "Dev"], cwd=repo) + (repo / "README.md").write_text("before\n", encoding="utf-8") + _run_git(["add", "README.md"], cwd=repo) + _run_git(["commit", "-m", "Commit present when the run starts"], cwd=repo) + sha_at_start = _run_git(["rev-parse", "HEAD"], cwd=repo) + + dataset_path = repo / "dataset.jsonl" + _write_dataset(dataset_path) + config_path = repo / "agentops.yaml" + _write_config(config_path, agent_url=good_url, dataset=dataset_path) + config = load_agentops_config(config_path) + + original_evaluate_row = orchestrator._evaluate_row + + def _evaluate_row_then_commit_mid_run(**kwargs): + row_result = original_evaluate_row(**kwargs) + if kwargs.get("index") == 0: + # Simulate the developer committing/checking out something + # else while this (slow, real-world) run is still executing. + (repo / "README.md").write_text("after\n", encoding="utf-8") + _run_git(["commit", "-am", "Made while the run was executing"], cwd=repo) + return row_result + + monkeypatch.setattr( + orchestrator, "_evaluate_row", _evaluate_row_then_commit_mid_run + ) + + result = run_evaluation( + config, + options=RunOptions( + config_path=config_path, + output_dir=tmp_path / "out", + timeout_seconds=10.0, + ), + ) + + sha_after_run = _run_git(["rev-parse", "HEAD"], cwd=repo) + assert sha_after_run != sha_at_start # the mid-run commit really happened + assert result.commit is not None + assert result.commit.sha == sha_at_start + assert result.commit.sha != sha_after_run From 8e28f0766be5a8c02ed1ea3170848df2135c71ec Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 11:57:55 -0300 Subject: [PATCH 25/30] fix(doctor): don't fabricate a cause when the attributed pair didn't regress --- src/agentops/agent/checks/regression.py | 64 +++++++++++++++------ tests/unit/test_agent_checks_regression.py | 67 ++++++++++++++++++++++ 2 files changed, 112 insertions(+), 19 deletions(-) diff --git a/src/agentops/agent/checks/regression.py b/src/agentops/agent/checks/regression.py index 422d7be7..15e1f446 100644 --- a/src/agentops/agent/checks/regression.py +++ b/src/agentops/agent/checks/regression.py @@ -11,7 +11,11 @@ from agentops.agent.findings import Category, Finding, Severity from agentops.agent.sources.results_history import ResultsHistory, RunSummary from agentops.core.results import RegressionInsight, RunResult -from agentops.pipeline.regression_insight import build_regression_insight, resolve_report_url +from agentops.pipeline.regression_insight import ( + build_regression_insight, + metric_improved, + resolve_report_url, +) def _load_run_result(summary: RunSummary) -> Optional[RunResult]: @@ -162,31 +166,53 @@ def run_regression_check( } insight: Optional[RegressionInsight] = None + reason: Optional[str] = None if latest_result is not None and previous_result is not None: - insight = build_regression_insight( - previous_result, - latest_result, - metrics=[metric], - workspace=workspace, - # Both are already-completed historical runs, so a sidecar - # `cloud_evaluation.json` next to either (from - # `execution: cloud` or a finished `publish: true`) is - # complete by now - unlike the in-flight `--baseline` - # comparison in `pipeline.comparison`. - from_report_url=resolve_report_url( - previous_result, results_path=previous_run.raw_path - ), - to_report_url=resolve_report_url(latest_result, results_path=latest.raw_path), - ) + # `previous_run` is selected independently of `baseline_runs` + # (by lineage, not fingerprint - see above), so the metric can + # have regressed against the rolling baseline while actually + # *improving* between this specific pair (e.g. a one-off + # excursion run sits between them with an unrelated, much + # lower value). Building an insight from a pair that doesn't + # itself show the drop would describe an improvement as the + # regression's cause - skip it and say so instead of + # fabricating a misleading explanation. + previous_value = previous_result.aggregate_metrics.get(metric) + latest_value = latest_result.aggregate_metrics.get(metric) + if metric_improved(metric, latest_value, previous_value) is False: + insight = build_regression_insight( + previous_result, + latest_result, + metrics=[metric], + workspace=workspace, + # Both are already-completed historical runs, so a sidecar + # `cloud_evaluation.json` next to either (from + # `execution: cloud` or a finished `publish: true`) is + # complete by now - unlike the in-flight `--baseline` + # comparison in `pipeline.comparison`. + from_report_url=resolve_report_url( + previous_result, results_path=previous_run.raw_path + ), + to_report_url=resolve_report_url( + latest_result, results_path=latest.raw_path + ), + ) + else: + reason = ( + "attribution unavailable: the immediately preceding " + "comparable run didn't show this drop - likely driven " + "by the broader rolling baseline instead" + ) if insight is not None: evidence["insight"] = insight.model_dump(mode="json") recommendation = insight.explanation if insight.suggested_action: recommendation = f"{recommendation} {insight.suggested_action}" else: - reason = _attribution_unavailable_reason( - latest, previous_run, latest_result, previous_result - ) + if reason is None: + reason = _attribution_unavailable_reason( + latest, previous_run, latest_result, previous_result + ) evidence["attribution_unavailable"] = reason recommendation = f"{recommendation} ({reason})" diff --git a/tests/unit/test_agent_checks_regression.py b/tests/unit/test_agent_checks_regression.py index d23d9f76..41f3cfc6 100644 --- a/tests/unit/test_agent_checks_regression.py +++ b/tests/unit/test_agent_checks_regression.py @@ -432,3 +432,70 @@ def test_first_run_after_a_version_bump_is_a_known_accepted_gap_in_doctor() -> N findings = run_regression_check(history, config) assert findings == [] + + +def test_no_insight_when_the_selected_pair_did_not_itself_regress(tmp_path) -> None: + """`previous_run` is selected independently of `baseline_runs` (by + lineage, not fingerprint), so a metric can regress against the + rolling baseline while actually *improving* between that specific + pair: v4=0.90, then a one-off v5 excursion at 0.50, then v4=0.70 again + - the rolling mean (0.90) vs 0.70 is a real ~22% drop that must still + fire, but the immediately preceding run (v5, 0.50) to latest (0.70) is + an *increase*, not a drop. Building an insight from that pair would + describe an improvement as the regression's cause - it must be + skipped, with a clear reason, instead. + """ + v4_path = tmp_path / "v4" / "results.json" + v5_path = tmp_path / "v5" / "results.json" + latest_path = tmp_path / "latest" / "results.json" + _write_result_json( + v4_path, accuracy=0.90, version="4", deployment="gpt-4o", commit_sha="1" * 40 + ) + _write_result_json( + v5_path, accuracy=0.50, version="5", deployment="gpt-4o", commit_sha="2" * 40 + ) + _write_result_json( + latest_path, accuracy=0.70, version="4", deployment="gpt-4o", commit_sha="3" * 40 + ) + + history = ResultsHistory( + runs=[ + _run( + {"accuracy": 0.90}, + run_id="v4", + offset_days=-2, + fingerprint="V4", + lineage_key="L", + raw_path=v4_path, + ), + _run( + {"accuracy": 0.50}, + run_id="v5", + offset_days=-1, + fingerprint="V5", + lineage_key="L", + raw_path=v5_path, + ), + _run( + {"accuracy": 0.70}, + run_id="latest", + offset_days=0, + fingerprint="V4", + lineage_key="L", + raw_path=latest_path, + ), + ] + ) + config = RegressionCheckConfig(metrics=["accuracy"], threshold_drop=0.10, min_runs=2) + + findings = run_regression_check(history, config) + + assert len(findings) == 1 + finding = findings[0] + assert "insight" not in finding.evidence + assert finding.evidence["attribution_unavailable"] == ( + "attribution unavailable: the immediately preceding comparable " + "run didn't show this drop - likely driven by the broader rolling " + "baseline instead" + ) + assert "didn't show this drop" in finding.recommendation From a282bf44590ccfab46a4b4fa2fa2cc007292bd6d Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 12:10:48 -0300 Subject: [PATCH 26/30] metrics improvements --- src/agentops/agent/checks/regression.py | 4 +- src/agentops/agent/cockpit.py | 2 + src/agentops/pipeline/comparison.py | 18 +++- src/agentops/pipeline/regression_insight.py | 102 ++++++++++++++++---- 4 files changed, 103 insertions(+), 23 deletions(-) diff --git a/src/agentops/agent/checks/regression.py b/src/agentops/agent/checks/regression.py index 15e1f446..5ee260fe 100644 --- a/src/agentops/agent/checks/regression.py +++ b/src/agentops/agent/checks/regression.py @@ -14,6 +14,7 @@ from agentops.pipeline.regression_insight import ( build_regression_insight, metric_improved, + metric_threshold_criteria, resolve_report_url, ) @@ -179,7 +180,8 @@ def run_regression_check( # fabricating a misleading explanation. previous_value = previous_result.aggregate_metrics.get(metric) latest_value = latest_result.aggregate_metrics.get(metric) - if metric_improved(metric, latest_value, previous_value) is False: + criteria = metric_threshold_criteria(metric, latest_result, previous_result) + if metric_improved(metric, latest_value, previous_value, criteria=criteria) is False: insight = build_regression_insight( previous_result, latest_result, diff --git a/src/agentops/agent/cockpit.py b/src/agentops/agent/cockpit.py index 5c3efa24..4d516afd 100644 --- a/src/agentops/agent/cockpit.py +++ b/src/agentops/agent/cockpit.py @@ -40,6 +40,7 @@ ) from agentops.core.results import RunResult from agentops.pipeline.comparison import LOWER_IS_BETTER_METRICS, metric_improved +from agentops.pipeline.regression_insight import metric_threshold_criteria from agentops.pipeline.regression_insight import build_changed_inputs from agentops.utils.yaml import load_yaml @@ -1009,6 +1010,7 @@ def _attach_version_history(runs: List[Dict[str, Any]]) -> None: m, current_full.aggregate_metrics[m], previous_full.aggregate_metrics[m], + criteria=metric_threshold_criteria(m, current_full, previous_full), ) is False ) diff --git a/src/agentops/pipeline/comparison.py b/src/agentops/pipeline/comparison.py index 7b06f51d..c83da4d1 100644 --- a/src/agentops/pipeline/comparison.py +++ b/src/agentops/pipeline/comparison.py @@ -16,6 +16,7 @@ LOWER_IS_BETTER_METRICS, build_regression_insight, metric_improved, + metric_threshold_criteria, resolve_report_url, ) @@ -34,8 +35,14 @@ def load_baseline(path: Path) -> RunResult: return RunResult.model_validate(payload) -def _direction(metric: str, current: Optional[float], baseline: Optional[float]) -> str: - improved = metric_improved(metric, current, baseline) +def _direction( + metric: str, + current: Optional[float], + baseline: Optional[float], + *, + criteria: Optional[str] = None, +) -> str: + improved = metric_improved(metric, current, baseline, criteria=criteria) if improved is None: return "unchanged" return "improved" if improved else "regressed" @@ -69,7 +76,12 @@ def build_comparison( current=current_value, baseline=baseline_value, delta=delta, - direction=_direction(name, current_value, baseline_value), + direction=_direction( + name, + current_value, + baseline_value, + criteria=metric_threshold_criteria(name, current, baseline), + ), ) ) diff --git a/src/agentops/pipeline/regression_insight.py b/src/agentops/pipeline/regression_insight.py index 20d9b484..2f149cc5 100644 --- a/src/agentops/pipeline/regression_insight.py +++ b/src/agentops/pipeline/regression_insight.py @@ -40,46 +40,105 @@ "thresholds": "threshold change", } -# Metrics where a lower value is the better outcome. Every other metric -# (and every evaluator's default threshold - see `core/evaluators.py`, -# where this is the only metric with a `<=` default) is treated as -# higher-is-better. Lives here (rather than `pipeline.comparison`, its only -# other consumer) because this module has no dependency on `comparison`, -# while `comparison` already depends on this module for -# `build_regression_insight` - putting it here avoids a circular import. +# Fallback for metrics with no threshold criteria recorded on either +# compared run - every evaluator's default threshold (see +# `core/evaluators.py`) is `>=` except this one (`<=`). Only a fallback: +# `agentops.yaml` lets *any* metric name declare a `<`/`<=` threshold +# (and `execution: azd` can introduce entirely custom metric names this +# way - see AGENTS.md), so a metric's own recorded `ThresholdEvaluation` +# criteria (see `_threshold_criteria`) is the real source of truth +# whenever it's available; this set only covers the common case where +# neither compared run happens to carry one. Lives here (rather than +# `pipeline.comparison`, its only other consumer) because this module has +# no dependency on `comparison`, while `comparison` already depends on +# this module for `build_regression_insight` - putting it here avoids a +# circular import. LOWER_IS_BETTER_METRICS = frozenset({"avg_latency_seconds"}) +def _threshold_criteria(run: RunResult, metric: str) -> Optional[str]: + """The ``<=``/``>=``/etc. operator configured for ``metric`` on ``run``, + from its own recorded ``ThresholdEvaluation`` list, if any.""" + for threshold in run.thresholds: + if threshold.metric == metric: + return threshold.criteria + return None + + +def metric_threshold_criteria( + metric: str, *runs: Optional[RunResult] +) -> Optional[str]: + """The first threshold criteria found for ``metric`` across ``runs``, + checked in order - callers typically pass the current/latest run + first, then the baseline/previous one, so either side recording a + threshold for this metric is enough to know its direction.""" + for run in runs: + if run is None: + continue + criteria = _threshold_criteria(run, metric) + if criteria is not None: + return criteria + return None + + +def _is_lower_is_better(metric: str, *, criteria: Optional[str] = None) -> bool: + """Whether a lower value is the better outcome for ``metric``. + + ``criteria`` (a recorded threshold operator like ``"<="``, from + ``metric_threshold_criteria``) wins when given - it reflects what this + specific metric was actually configured to mean, unlike + ``LOWER_IS_BETTER_METRICS``, which is only a fallback guess for the one + built-in metric known to commonly be lower-is-better. + """ + if criteria is not None: + if criteria.startswith("<"): + return True + if criteria.startswith(">"): + return False + return metric in LOWER_IS_BETTER_METRICS + + def metric_improved( - metric: str, current: Optional[float], baseline: Optional[float] + metric: str, + current: Optional[float], + baseline: Optional[float], + *, + criteria: Optional[str] = None, ) -> Optional[bool]: """Whether ``current`` is a better outcome than ``baseline`` for ``metric``. Returns ``None`` when the values are missing or equal (no judgement to make), ``True`` when ``current`` is the better value, ``False`` when - it's worse - accounting for ``metric`` being lower-is-better (e.g. - latency) vs. the higher-is-better default. + it's worse - accounting for ``metric``'s direction (see + ``_is_lower_is_better``). """ if current is None or baseline is None or current == baseline: return None - if metric in LOWER_IS_BETTER_METRICS: + if _is_lower_is_better(metric, criteria=criteria): return current < baseline return current > baseline -def regression_severity(metric: str, from_value: float, to_value: float) -> float: +def regression_severity( + metric: str, + from_value: float, + to_value: float, + *, + criteria: Optional[str] = None, +) -> float: """How severely ``metric`` regressed from ``from_value`` to ``to_value``, as a fraction of ``from_value`` when that's meaningful, else as an absolute change. Used only to order ``RegressionInsight.regressed_metrics`` - worst-first - direction-aware, so a lower-is-better metric (e.g. - latency) getting worse produces a positive value just like any other - regression. A ``from_value`` at or below zero can't produce a - meaningful ratio, so the absolute change is used instead of collapsing - to zero. + worst-first - direction-aware (see ``_is_lower_is_better``), so a + lower-is-better metric (e.g. latency, or any custom metric with a + ``<=``/``<`` threshold) getting worse produces a positive value just + like any other regression. A ``from_value`` at or below zero can't + produce a meaningful ratio, so the absolute change is used instead of + collapsing to zero. """ worse_by = ( to_value - from_value - if metric in LOWER_IS_BETTER_METRICS + if _is_lower_is_better(metric, criteria=criteria) else from_value - to_value ) if worse_by <= 0: @@ -307,7 +366,12 @@ def build_regression_insight( if not regressed_metrics: return None regressed_metrics.sort( - key=lambda m: regression_severity(m.metric, m.from_value, m.to_value), + key=lambda m: regression_severity( + m.metric, + m.from_value, + m.to_value, + criteria=metric_threshold_criteria(m.metric, to_run, from_run), + ), reverse=True, ) From d7035f4189d39b384c6c64f7595ff90fa1ddd4ca Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 15:06:29 -0300 Subject: [PATCH 27/30] fix(cockpit,doctor): prefer url over name in lineage identity --- src/agentops/agent/cockpit.py | 29 ++++++++++++------ src/agentops/agent/sources/results_history.py | 30 +++++++++++++------ 2 files changed, 41 insertions(+), 18 deletions(-) diff --git a/src/agentops/agent/cockpit.py b/src/agentops/agent/cockpit.py index 4d516afd..78cdebc1 100644 --- a/src/agentops/agent/cockpit.py +++ b/src/agentops/agent/cockpit.py @@ -926,16 +926,27 @@ def _version_lineage_key(data: Dict[str, Any]) -> Optional[str]: """ raw_target = data.get("target") target: Dict[str, Any] = raw_target if isinstance(raw_target, dict) else {} - # `raw` is the identity fallback for every target kind except - # `model_direct` (`model:`), where `raw` itself embeds the - # deployment (e.g. "model:gpt-4o" vs "model:gpt-4o-mini") - using it - # there would make a deployment change start a new lineage instead of - # being detected as a change within the same one, defeating the - # version/deployment-blind point of this key. `kind` has no such - # problem: it's the constant "model_direct" for every run of this kind. + # `url` wins over `name` when both are present: for `foundry_hosted`, + # `name` is only the agent-name fragment parsed *out of* that same + # URL (see TargetInfo's docstring), so two different Foundry projects + # with a same-named agent would otherwise collide on an identical, + # project-blind `name` and get merged into one lineage - `url` already + # carries the full, project-qualified identity. `raw` is the fallback + # for every remaining kind except `model_direct` (`model:`), + # where `raw` itself embeds the deployment (e.g. "model:gpt-4o" vs + # "model:gpt-4o-mini") - using it there would make a deployment change + # start a new lineage instead of being detected as a change within the + # same one. `kind` has no such problem: it's the constant + # "model_direct" for every run of this kind. + # + # Known gap, not fixed here: `foundry_prompt` has neither `url` nor any + # other project-qualified field - just `name`/`version` - so two + # different projects with a same-named prompt agent still collide. + # Fixing that needs a new field capturing the project endpoint on + # TargetInfo for prompt agents, which is a bigger, separate change. agent_identity = ( - target.get("name") - or target.get("url") + target.get("url") + or target.get("name") or ( target.get("kind") if target.get("kind") == "model_direct" diff --git a/src/agentops/agent/sources/results_history.py b/src/agentops/agent/sources/results_history.py index babe1179..c422a347 100644 --- a/src/agentops/agent/sources/results_history.py +++ b/src/agentops/agent/sources/results_history.py @@ -165,16 +165,28 @@ def _lineage_key(data: Dict[str, Any]) -> Optional[str]: """ raw_target = data.get("target") target: Dict[str, Any] = raw_target if isinstance(raw_target, dict) else {} - # `raw` is the identity fallback for every target kind except - # `model_direct` (`model:`), where `raw` itself embeds the - # deployment (e.g. "model:gpt-4o" vs "model:gpt-4o-mini") - using it - # there would make a deployment change look like a different agent - # entirely, defeating the version/deployment-blind point of this key. - # `kind` has no such problem: it's the constant "model_direct" for - # every run of this kind. + # `url` wins over `name` when both are present: for `foundry_hosted`, + # `name` is only the agent-name fragment parsed *out of* that same + # URL, so two different Foundry projects with a same-named agent would + # otherwise collide on an identical, project-blind `name` and Doctor + # could pick a run from the wrong project as the "previous comparable + # run" - `url` already carries the full, project-qualified identity. + # `raw` is the fallback for every remaining kind except `model_direct` + # (`model:`), where `raw` itself embeds the deployment + # (e.g. "model:gpt-4o" vs "model:gpt-4o-mini") - using it there would + # make a deployment change look like a different agent entirely, + # defeating the version/deployment-blind point of this key. `kind` has + # no such problem: it's the constant "model_direct" for every run of + # this kind. + # + # Known gap, not fixed here: `foundry_prompt` has neither `url` nor + # any other project-qualified field - just `name`/`version` - so two + # different projects with a same-named prompt agent still collide. + # Fixing that needs a new field capturing the project endpoint on + # TargetInfo for prompt agents, which is a bigger, separate change. agent_identity = ( - target.get("name") - or target.get("url") + target.get("url") + or target.get("name") or ( target.get("kind") if target.get("kind") == "model_direct" From ca026eb834aca8913bb0af63ec6371e304389cbc Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 15:06:55 -0300 Subject: [PATCH 28/30] fix(pipeline): require same lineage before causal attribution, not just commits --- src/agentops/pipeline/regression_insight.py | 43 +++++++++- .../test_regression_commit_attribution.py | 78 +++++++++++++++---- tests/unit/test_pipeline_comparison.py | 42 +++++++++- tests/unit/test_regression_insight.py | 42 ++++++++++ 4 files changed, 186 insertions(+), 19 deletions(-) diff --git a/src/agentops/pipeline/regression_insight.py b/src/agentops/pipeline/regression_insight.py index 2f149cc5..3a37b57e 100644 --- a/src/agentops/pipeline/regression_insight.py +++ b/src/agentops/pipeline/regression_insight.py @@ -27,6 +27,7 @@ RegressedMetric, RegressionInsight, RunResult, + TargetInfo, ) from agentops.pipeline.commit_info import commit_exists_locally @@ -318,6 +319,42 @@ def _suggested_action(changed_inputs: List[ChangedInput]) -> Optional[str]: return f"Review the {_join_with_and(labels)}; consider reverting one at a time to isolate the cause." +def _agent_identity(target: TargetInfo) -> Optional[str]: + """Mirrors ``cockpit._version_lineage_key``'s / ``results_history._lineage_key``'s + dict-based ``agent_identity`` resolution, for a ``RunResult.target`` + (``TargetInfo``) directly - see those functions' docstrings for why + ``url`` wins over ``name``, and why ``model_direct`` uses ``kind`` + rather than ``raw``. + """ + if target.url: + return target.url + if target.name: + return target.name + if target.kind == "model_direct": + return target.kind + return target.raw + + +def same_lineage(from_run: RunResult, to_run: RunResult) -> bool: + """Whether two runs are comparable for causal regression attribution + (FR-005): same dataset, evaluator set, and agent identity - but + deliberately version/deployment-blind, matching the lineage rule + Doctor's own ``previous_run`` selection already enforces by + construction (see ``agent.checks.regression``). An explicit + ``--baseline`` file has no such guarantee built in (the user can point + it at any ``results.json``), so ``build_regression_insight`` checks + this itself rather than trusting every caller to have already + filtered for it - comparing two runs of different agents, datasets, or + evaluator sets could otherwise produce a "likely cause" that isn't + real. + """ + if from_run.dataset_path != to_run.dataset_path: + return False + if sorted(from_run.evaluators) != sorted(to_run.evaluators): + return False + return _agent_identity(from_run.target) == _agent_identity(to_run.target) + + def build_regression_insight( from_run: RunResult, to_run: RunResult, @@ -337,7 +374,9 @@ def build_regression_insight( metrics move together. Returns ``None`` (produces no fabricated cause) when either run lacks - commit metadata, or when none of ``metrics`` has a value on both runs, + commit metadata, when the two runs aren't comparable in the first + place (``same_lineage`` - FR-005: different agent, dataset, or + evaluator set), or when none of ``metrics`` has a value on both runs, per FR-009 - an individual metric missing a value on either side is just skipped rather than failing the whole insight. Otherwise diffs the runs' recorded fields (see ``build_changed_inputs``) and renders a @@ -353,6 +392,8 @@ def build_regression_insight( """ if from_run.commit is None or to_run.commit is None: return None + if not same_lineage(from_run, to_run): + return None regressed_metrics: List[RegressedMetric] = [] for metric in metrics: diff --git a/tests/integration/test_regression_commit_attribution.py b/tests/integration/test_regression_commit_attribution.py index 7ac9904c..24ba0e5a 100644 --- a/tests/integration/test_regression_commit_attribution.py +++ b/tests/integration/test_regression_commit_attribution.py @@ -68,6 +68,37 @@ def log_message(self, *args, **kwargs) -> None: # noqa: D401 pass +class _ToggleHandler(BaseHTTPRequestHandler): + """Answers correctly or incorrectly depending on a shared class-level + flag, so the *same* agent URL (and dataset) can simulate "this agent's + behavior changed between two commits" without actually changing + identity - build_regression_insight now requires the compared runs' + agent/dataset/evaluators to match (FR-005 comparability) before + producing a causal insight, which a real `--baseline` comparison + against a different dataset/endpoint correctly no longer does. + """ + + answer_correctly = True + + def do_POST(self) -> None: # noqa: N802 + length = int(self.headers.get("Content-Length", "0")) + body = json.loads(self.rfile.read(length).decode("utf-8")) + if type(self).answer_correctly: + message = body.get("message", "") + answer = _EXACT_ANSWERS.get(message, "") + else: + answer = "completely unrelated response" + payload = json.dumps({"text": answer}).encode("utf-8") + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(payload))) + self.end_headers() + self.wfile.write(payload) + + def log_message(self, *args, **kwargs) -> None: # noqa: D401 + pass + + def _serve(handler_cls): server = HTTPServer(("127.0.0.1", 0), handler_cls) thread = threading.Thread(target=server.serve_forever, daemon=True) @@ -118,15 +149,26 @@ def good_and_bad_servers(): bad_thread.join(timeout=1) -def _run_baseline_then_regressed(tmp_path: Path, monkeypatch, good_url: str, bad_url: str): - """Runs a good then a regressed evaluation, with commit capture mocked. +@pytest.fixture() +def toggle_server(): + _ToggleHandler.answer_correctly = True + server, thread, url = _serve(_ToggleHandler) + try: + yield url + finally: + server.shutdown() + thread.join(timeout=1) + + +def _run_baseline_then_regressed(tmp_path: Path, monkeypatch, agent_url: str): + """Runs a good then a regressed evaluation against the *same* agent + URL and dataset (only the toggle server's behavior differs between + them), with commit capture mocked. Returns (baseline_result, current_result, current_dir). """ - baseline_dataset = tmp_path / "dataset-v1.jsonl" - current_dataset = tmp_path / "dataset-v2.jsonl" - _write_dataset(baseline_dataset) - _write_dataset(current_dataset) + dataset_path = tmp_path / "dataset.jsonl" + _write_dataset(dataset_path) commits = iter([_fake_commit("a" * 40), _fake_commit("b" * 40)]) monkeypatch.setattr( @@ -134,9 +176,10 @@ def _run_baseline_then_regressed(tmp_path: Path, monkeypatch, good_url: str, bad ) baseline_config_path = tmp_path / "agentops-baseline.yaml" - _write_config(baseline_config_path, agent_url=good_url, dataset=baseline_dataset) + _write_config(baseline_config_path, agent_url=agent_url, dataset=dataset_path) baseline_config = load_agentops_config(baseline_config_path) + _ToggleHandler.answer_correctly = True baseline_dir = tmp_path / "baseline" baseline_result = run_evaluation( baseline_config, @@ -148,9 +191,10 @@ def _run_baseline_then_regressed(tmp_path: Path, monkeypatch, good_url: str, bad ) current_config_path = tmp_path / "agentops-current.yaml" - _write_config(current_config_path, agent_url=bad_url, dataset=current_dataset) + _write_config(current_config_path, agent_url=agent_url, dataset=dataset_path) current_config = load_agentops_config(current_config_path) + _ToggleHandler.answer_correctly = False current_dir = tmp_path / "current" current_result = run_evaluation( current_config, @@ -166,12 +210,10 @@ def _run_baseline_then_regressed(tmp_path: Path, monkeypatch, good_url: str, bad def test_regression_commit_attribution_end_to_end( - tmp_path: Path, monkeypatch, good_and_bad_servers + tmp_path: Path, monkeypatch, toggle_server ) -> None: - good_url, bad_url = good_and_bad_servers - baseline_result, current_result, current_dir = _run_baseline_then_regressed( - tmp_path, monkeypatch, good_url, bad_url + tmp_path, monkeypatch, toggle_server ) assert baseline_result.aggregate_metrics["f1_score"] == pytest.approx(1.0) @@ -186,7 +228,11 @@ def test_regression_commit_attribution_end_to_end( assert len(insight.regressed_metrics) == 1 assert insight.regressed_metrics[0].metric == "f1_score" assert insight.regressed_metrics[0].from_value == pytest.approx(1.0) - assert any(c.field == "dataset" for c in insight.changed_inputs) + # Same agent/dataset/evaluators on both sides (FR-005 comparability) - + # only the toggle server's behavior differs, which build_changed_inputs + # has no way to see, so there is no tracked "changed input" here; the + # explanation still names the metric's before/after values. + assert insight.changed_inputs == [] report_text = (current_dir / "report.md").read_text(encoding="utf-8") assert "## Regression Insight" in report_text @@ -199,16 +245,14 @@ def test_regression_commit_attribution_end_to_end( def test_regression_commit_attribution_purely_local( - tmp_path: Path, monkeypatch, good_and_bad_servers + tmp_path: Path, monkeypatch, toggle_server ) -> None: """The same outcome holds with no CI env vars - commit resolved via local git.""" - good_url, bad_url = good_and_bad_servers - for env_var in ("GITHUB_SHA", "BUILD_SOURCEVERSION", "Build.SourceVersion"): monkeypatch.delenv(env_var, raising=False) _baseline_result, current_result, current_dir = _run_baseline_then_regressed( - tmp_path, monkeypatch, good_url, bad_url + tmp_path, monkeypatch, toggle_server ) # Commit capture itself is mocked here (as in the CI-style test above) to diff --git a/tests/unit/test_pipeline_comparison.py b/tests/unit/test_pipeline_comparison.py index 1a6032e3..8337fd45 100644 --- a/tests/unit/test_pipeline_comparison.py +++ b/tests/unit/test_pipeline_comparison.py @@ -4,7 +4,13 @@ from pathlib import Path -from agentops.core.results import CommitInfo, RunResult, RunSummary, TargetInfo +from agentops.core.results import ( + CommitInfo, + RunResult, + RunSummary, + TargetInfo, + ThresholdEvaluation, +) from agentops.pipeline import comparison @@ -147,6 +153,40 @@ def test_latency_increase_is_regressed_and_explained(): assert info.insight.regressed_metrics[0].metric == "avg_latency_seconds" +def _with_threshold(run: RunResult, *, metric: str, criteria: str) -> RunResult: + run.thresholds = [ + ThresholdEvaluation(metric=metric, criteria=criteria, expected="", actual="", passed=True) + ] + return run + + +def test_custom_lower_is_better_metric_drop_is_improved_not_regressed(): + """``LOWER_IS_BETTER_METRICS`` only covers the one built-in metric known + to commonly be lower-is-better (latency) - any other metric name can + still be configured as lower-is-better via an explicit `<=`/`<` + threshold in agentops.yaml (or a custom metric execution: azd imports). + A drop in such a metric must be read from its own recorded threshold + criteria, not assumed higher-is-better by default.""" + baseline = _with_threshold( + _run(accuracy=0.91, commit=_commit("a" * 40)), + metric="error_rate", + criteria="<=", + ) + baseline.aggregate_metrics["error_rate"] = 0.10 + current = _with_threshold( + _run(accuracy=0.91, commit=_commit("b" * 40)), metric="error_rate", criteria="<=" + ) + current.aggregate_metrics["error_rate"] = 0.04 + + info = comparison.build_comparison( + current=current, baseline=baseline, baseline_path=Path(".agentops/baseline/results.json") + ) + + error_rate_metric = next(m for m in info.metrics if m.metric == "error_rate") + assert error_rate_metric.direction == "improved" + assert info.insight is None + + def test_insight_carries_baseline_report_url_from_sidecar_file(tmp_path: Path): """``baseline`` is a prior, fully-published run - a sidecar ``cloud_evaluation.json`` next to its ``results.json`` (as a completed diff --git a/tests/unit/test_regression_insight.py b/tests/unit/test_regression_insight.py index 817629d8..a2c9ada9 100644 --- a/tests/unit/test_regression_insight.py +++ b/tests/unit/test_regression_insight.py @@ -156,6 +156,48 @@ def test_no_insight_when_to_commit_missing(): assert regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) is None +def test_no_insight_when_agent_identity_differs(): + """FR-005: two runs of different agents aren't comparable for causal + attribution, even with commit metadata and an apparent metric drop - + an explicit --baseline file has no guarantee it points at the same + agent, unlike Doctor's own previous_run selection (lineage-filtered by + construction).""" + from_run = _run(name="greeter", accuracy=0.91, commit=_commit("a" * 40)) + to_run = _run(name="other-agent", accuracy=0.79, commit=_commit("b" * 40)) + + assert regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) is None + + +def test_no_insight_when_dataset_differs(): + from_run = _run(dataset_path="data/a.jsonl", accuracy=0.91, commit=_commit("a" * 40)) + to_run = _run(dataset_path="data/b.jsonl", accuracy=0.79, commit=_commit("b" * 40)) + + assert regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) is None + + +def test_no_insight_when_evaluators_differ(): + from_run = _run( + evaluators=["CoherenceEvaluator"], accuracy=0.91, commit=_commit("a" * 40) + ) + to_run = _run( + evaluators=["CoherenceEvaluator", "FluencyEvaluator"], + accuracy=0.79, + commit=_commit("b" * 40), + ) + + assert regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) is None + + +def test_same_lineage_is_version_blind(): + """A version/deployment bump alone must not count as a lineage + mismatch - that's exactly the change this feature exists to attribute + a regression to.""" + from_run = _run(version="3", deployment="gpt-4o", commit=_commit("a" * 40)) + to_run = _run(version="4", deployment="gpt-4o-mini", commit=_commit("b" * 40)) + + assert regression_insight.same_lineage(from_run, to_run) is True + + def test_no_insight_when_metric_missing_on_either_run(): from_run = _run(commit=_commit("a" * 40)) to_run = _run(commit=_commit("b" * 40)) From 7fffbfe89c6c55a84009bb807ce181abec39f781 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 16:02:11 -0300 Subject: [PATCH 29/30] fix(pipeline): normalize hosted-agent URLs for lineage; attribute --- .../012-regression-commit-attribution/spec.md | 4 +- src/agentops/agent/cockpit.py | 40 ++----- src/agentops/agent/sources/results_history.py | 35 ++---- src/agentops/pipeline/regression_insight.py | 112 +++++++++++++----- tests/unit/test_regression_insight.py | 60 +++++++++- 5 files changed, 166 insertions(+), 85 deletions(-) diff --git a/specs/012-regression-commit-attribution/spec.md b/specs/012-regression-commit-attribution/spec.md index e86b1f1e..cf8bfcd2 100644 --- a/specs/012-regression-commit-attribution/spec.md +++ b/specs/012-regression-commit-attribution/spec.md @@ -76,7 +76,7 @@ A developer running `agentops eval run` locally against a git-managed workspace - **FR-002**: For evaluation runs against a Foundry hosted agent executed via cloud or azd execution in a CI environment, the system MUST reliably capture commit metadata from CI-provided source-control information, at the same reliability level already achieved for existing CI-based commit capture. - **FR-003**: For local execution, the system MUST make a best-effort attempt to capture commit metadata from the local git repository when the working directory is one, and MUST allow the run to complete normally, without error, when it is not. - **FR-004**: Commit metadata MUST be persisted as an additive part of the run's stored result such that existing consumers of that result format that are unaware of the new field are unaffected. -- **FR-005**: The system MUST only compare two runs for regression-cause attribution when they share the same dataset and evaluator set, and the same agent identity — but deliberately *independent of* the agent's version/model/deployment, since a change in exactly those fields is the attribution this feature exists to surface. This is a coarser, version-blind comparability rule distinct from the stricter, version-inclusive methodology fingerprint already used for the rolling-baseline drop-percentage calculation and other existing checks (e.g. flaky-metric detection), where mixing versions would be genuinely spurious. +- **FR-005**: The system MUST only compare two runs for regression-cause attribution when they share the same agent identity — but deliberately *independent of* the agent's version/model/deployment, since a change in exactly those fields is the attribution this feature exists to surface. This comparability rule deliberately does **not** require the same dataset or evaluator set: a dataset or evaluator-set change between the two runs is itself one of the things FR-006 requires attributing a regression to (see `build_changed_inputs`'s dataset/evaluator-change detection), so blocking on it would make that detection unreachable; it would also compare an absolute, checkout-resolved dataset path (see FR-003's "local execution" note), which a committed baseline produced in a different checkout than the current run would almost never match even for the literal same repository-relative dataset. This agent-identity rule is distinct from the stricter, version-inclusive methodology fingerprint already used for the rolling-baseline drop-percentage calculation and other existing checks (e.g. flaky-metric detection), where mixing versions would be genuinely spurious - and from FR-012's lineage key (which groups by dataset/evaluator/agent identity together, for Cockpit's own display grouping). - **FR-006**: When a regression is detected between two comparable runs that both have recorded commit metadata, the system MUST determine, at minimum, whether the evaluated agent's system prompt content, its model/deployment identifier, and other tracked run configuration (dataset, evaluator set, thresholds) changed between the two commits. - **FR-007**: When a regression cause is determined, the system MUST produce a plain-language, one-to-two-sentence explanation naming the regressed metric's before/after values and the concrete inputs found to have changed. - **FR-008**: Where a likely cause is identified, the system SHOULD include a brief suggested corrective adjustment tied directly to that cause. @@ -106,7 +106,7 @@ A developer running `agentops eval run` locally against a git-managed workspace - "Hosted agents" refers to Foundry hosted agent targets evaluated via cloud or azd execution, consistent with how the system already classifies agent targets. - The CI environment used for the PR pipeline provides the commit SHA of the code under evaluation through standard CI environment variables, the same class of information already relied on for existing CI-based commit capture elsewhere in the system. -- "Comparable runs" for regression-cause attribution (FR-005) and Cockpit's history view (FR-012) means a coarser, version/deployment-blind variant of the existing methodology-fingerprint concept: same dataset, evaluator set, and agent identity, but *not* requiring the same prompt version/model/deployment - the opposite of the stricter, version-inclusive fingerprint already used for the rolling-baseline drop-percentage math and other existing checks. This feature does introduce this new, coarser comparability rule specifically because the stricter one would exclude the very version/deployment change this feature exists to attribute. +- "Comparable runs" for Cockpit's history view (FR-012) means a coarser, version/deployment-blind variant of the existing methodology-fingerprint concept: same dataset, evaluator set, and agent identity, but *not* requiring the same prompt version/model/deployment - the opposite of the stricter, version-inclusive fingerprint already used for the rolling-baseline drop-percentage math and other existing checks. For regression-cause attribution specifically (FR-005), "comparable" is narrower still in what it requires (agent identity only, not dataset/evaluators - see FR-005) since attributing a dataset/evaluator change is itself in scope there. This feature does introduce both of these new comparability rules specifically because the stricter, version-inclusive fingerprint would exclude the very version/deployment change this feature exists to attribute. - A full commit-to-commit diff assumes both compared commits are reachable in the local git history available at comparison time; when they are not (e.g. a shallow clone), the system falls back to comparing the fields already recorded in each run's stored result instead of requiring direct git access to both commits. - "System prompt" content for diffing is whatever prompt content the run's configuration already records for the evaluated agent version; this feature does not introduce new prompt-capture beyond what configuration/version resolution already provides. - Suggested corrective adjustments are short and rule-based, directly tied to the specific change detected (e.g. a changed prompt suggests reviewing that prompt change); this is not a general-purpose troubleshooting assistant and does not call out to a language model. diff --git a/src/agentops/agent/cockpit.py b/src/agentops/agent/cockpit.py index 78cdebc1..d12364a8 100644 --- a/src/agentops/agent/cockpit.py +++ b/src/agentops/agent/cockpit.py @@ -40,7 +40,10 @@ ) from agentops.core.results import RunResult from agentops.pipeline.comparison import LOWER_IS_BETTER_METRICS, metric_improved -from agentops.pipeline.regression_insight import metric_threshold_criteria +from agentops.pipeline.regression_insight import ( + agent_identity_from_fields, + metric_threshold_criteria, +) from agentops.pipeline.regression_insight import build_changed_inputs from agentops.utils.yaml import load_yaml @@ -926,32 +929,15 @@ def _version_lineage_key(data: Dict[str, Any]) -> Optional[str]: """ raw_target = data.get("target") target: Dict[str, Any] = raw_target if isinstance(raw_target, dict) else {} - # `url` wins over `name` when both are present: for `foundry_hosted`, - # `name` is only the agent-name fragment parsed *out of* that same - # URL (see TargetInfo's docstring), so two different Foundry projects - # with a same-named agent would otherwise collide on an identical, - # project-blind `name` and get merged into one lineage - `url` already - # carries the full, project-qualified identity. `raw` is the fallback - # for every remaining kind except `model_direct` (`model:`), - # where `raw` itself embeds the deployment (e.g. "model:gpt-4o" vs - # "model:gpt-4o-mini") - using it there would make a deployment change - # start a new lineage instead of being detected as a change within the - # same one. `kind` has no such problem: it's the constant - # "model_direct" for every run of this kind. - # - # Known gap, not fixed here: `foundry_prompt` has neither `url` nor any - # other project-qualified field - just `name`/`version` - so two - # different projects with a same-named prompt agent still collide. - # Fixing that needs a new field capturing the project endpoint on - # TargetInfo for prompt agents, which is a bigger, separate change. - agent_identity = ( - target.get("url") - or target.get("name") - or ( - target.get("kind") - if target.get("kind") == "model_direct" - else target.get("raw") - ) + # See `regression_insight.agent_identity_from_fields`'s docstring - + # the single source of truth for this, shared with Doctor + # (results_history._lineage_key) and build_regression_insight's own + # comparability check, so the three can't drift apart again. + agent_identity = agent_identity_from_fields( + name=target.get("name"), + url=target.get("url"), + kind=target.get("kind"), + raw=target.get("raw"), ) dataset_path = data.get("dataset_path") evaluators_raw = data.get("evaluators") diff --git a/src/agentops/agent/sources/results_history.py b/src/agentops/agent/sources/results_history.py index c422a347..4818a8fc 100644 --- a/src/agentops/agent/sources/results_history.py +++ b/src/agentops/agent/sources/results_history.py @@ -17,6 +17,7 @@ from typing import Any, Dict, Iterable, List, Optional from agentops.agent.config import FoundryControlSourceConfig, ResultsHistorySourceConfig +from agentops.pipeline.regression_insight import agent_identity_from_fields log = logging.getLogger(__name__) @@ -165,32 +166,16 @@ def _lineage_key(data: Dict[str, Any]) -> Optional[str]: """ raw_target = data.get("target") target: Dict[str, Any] = raw_target if isinstance(raw_target, dict) else {} - # `url` wins over `name` when both are present: for `foundry_hosted`, - # `name` is only the agent-name fragment parsed *out of* that same - # URL, so two different Foundry projects with a same-named agent would - # otherwise collide on an identical, project-blind `name` and Doctor - # could pick a run from the wrong project as the "previous comparable - # run" - `url` already carries the full, project-qualified identity. - # `raw` is the fallback for every remaining kind except `model_direct` - # (`model:`), where `raw` itself embeds the deployment - # (e.g. "model:gpt-4o" vs "model:gpt-4o-mini") - using it there would - # make a deployment change look like a different agent entirely, - # defeating the version/deployment-blind point of this key. `kind` has - # no such problem: it's the constant "model_direct" for every run of - # this kind. - # - # Known gap, not fixed here: `foundry_prompt` has neither `url` nor - # any other project-qualified field - just `name`/`version` - so two - # different projects with a same-named prompt agent still collide. - # Fixing that needs a new field capturing the project endpoint on - # TargetInfo for prompt agents, which is a bigger, separate change. + # See `regression_insight.agent_identity_from_fields`'s docstring - the + # single source of truth for this, shared with Cockpit + # (cockpit._version_lineage_key) and build_regression_insight's own + # comparability check, so the three can't drift apart again. agent_identity = ( - target.get("url") - or target.get("name") - or ( - target.get("kind") - if target.get("kind") == "model_direct" - else target.get("raw") + agent_identity_from_fields( + name=target.get("name"), + url=target.get("url"), + kind=target.get("kind"), + raw=target.get("raw"), ) or (data.get("config") or {}).get("agent") ) diff --git a/src/agentops/pipeline/regression_insight.py b/src/agentops/pipeline/regression_insight.py index 3a37b57e..680ded9a 100644 --- a/src/agentops/pipeline/regression_insight.py +++ b/src/agentops/pipeline/regression_insight.py @@ -19,6 +19,7 @@ from __future__ import annotations import json +import re from pathlib import Path from typing import Any, Dict, List, Optional @@ -319,39 +320,96 @@ def _suggested_action(changed_inputs: List[ChangedInput]) -> Optional[str]: return f"Review the {_join_with_and(labels)}; consider reverting one at a time to isolate the cause." -def _agent_identity(target: TargetInfo) -> Optional[str]: - """Mirrors ``cockpit._version_lineage_key``'s / ``results_history._lineage_key``'s - dict-based ``agent_identity`` resolution, for a ``RunResult.target`` - (``TargetInfo``) directly - see those functions' docstrings for why - ``url`` wins over ``name``, and why ``model_direct`` uses ``kind`` - rather than ``raw``. +# Matches the `/versions/` segment a Foundry hosted-agent URL +# carries (see `core.agentops_config._HOSTED_AGENT_REFERENCE_RE` - the +# same pattern, kept independent since `core/` must stay free of +# `pipeline`/`agent` imports). A hosted agent's `url` is otherwise the +# strongest identity signal available (project + agent name, already +# qualified) - but it's version-specific, so used as-is it would make +# every version bump start a new lineage, exactly the change this feature +# exists to compare across. +_HOSTED_AGENT_VERSION_RE = re.compile(r"/versions/[^/?#]+", re.IGNORECASE) + + +def normalize_agent_url(url: str) -> str: + """Strip the ``/versions/`` segment from a Foundry hosted + agent URL, keeping the project+agent-qualified rest of it intact.""" + return _HOSTED_AGENT_VERSION_RE.sub("", url) + + +def agent_identity_from_fields( + *, + name: Optional[str], + url: Optional[str], + kind: Optional[str], + raw: Optional[str], +) -> Optional[str]: + """Shared agent-identity resolution for lineage grouping - the single + source of truth used by ``cockpit._version_lineage_key``, + ``results_history._lineage_key``, and this module's + ``_agent_identity`` (``RunResult``-based), so the three can't drift + apart again. + + ``url`` wins over ``name`` when both are present: for + ``foundry_hosted``, ``name`` is only the agent-name fragment parsed + *out of* that same URL (see ``TargetInfo``'s docstring), so two + different Foundry projects with a same-named agent would otherwise + collide on an identical, project-blind ``name``. The URL itself is + normalized (``normalize_agent_url``) to strip its version segment, so + a version bump doesn't start a new lineage - the opposite of using + ``name`` alone, which is already version-blind. + + ``raw`` is the fallback for every remaining kind except + ``model_direct`` (``model:``), where ``raw`` itself embeds + the deployment (e.g. ``"model:gpt-4o"`` vs ``"model:gpt-4o-mini"``) - + using it there would make a deployment change look like a different + agent entirely. ``kind`` has no such problem: it's the constant + ``"model_direct"`` for every run of this kind. + + Known gap, not fixed here: ``foundry_prompt`` has neither ``url`` nor + any other project-qualified field - just ``name``/``version`` - so two + different projects with a same-named prompt agent still collide. + Fixing that needs a new field capturing the project endpoint on + ``TargetInfo`` for prompt agents, which is a bigger, separate change. """ - if target.url: - return target.url - if target.name: - return target.name - if target.kind == "model_direct": - return target.kind - return target.raw + if url: + return normalize_agent_url(url) + if name: + return name + if kind == "model_direct": + return kind + return raw + + +def _agent_identity(target: TargetInfo) -> Optional[str]: + """``agent_identity_from_fields`` for a ``RunResult.target`` directly.""" + return agent_identity_from_fields( + name=target.name, url=target.url, kind=target.kind, raw=target.raw + ) def same_lineage(from_run: RunResult, to_run: RunResult) -> bool: """Whether two runs are comparable for causal regression attribution - (FR-005): same dataset, evaluator set, and agent identity - but - deliberately version/deployment-blind, matching the lineage rule - Doctor's own ``previous_run`` selection already enforces by - construction (see ``agent.checks.regression``). An explicit - ``--baseline`` file has no such guarantee built in (the user can point - it at any ``results.json``), so ``build_regression_insight`` checks - this itself rather than trusting every caller to have already - filtered for it - comparing two runs of different agents, datasets, or - evaluator sets could otherwise produce a "likely cause" that isn't - real. + (FR-005): same agent identity - but deliberately version/deployment- + blind, matching the lineage rule Doctor's own ``previous_run`` + selection already enforces by construction (see + ``agent.checks.regression``). An explicit ``--baseline`` file has no + such guarantee built in (the user can point it at any + ``results.json``), so ``build_regression_insight`` checks this itself + rather than trusting every caller to have already filtered for it - + comparing two runs of entirely different agents could otherwise + produce a "likely cause" that isn't real. + + Deliberately does *not* require the same ``dataset_path`` or + ``evaluators``: those are exactly the kind of change this feature + must still be able to attribute a regression to (FR-006, and + ``build_changed_inputs``'s own dataset/evaluator-change detection) - + blocking on them would make that detection unreachable. ``dataset_path`` + is also not checkout-stable (local runs persist an absolute, resolved + path - see ``services.dataset_source``), so two runs of the very same + repo-relative dataset produced from different checkouts (e.g. a + committed baseline vs. a fresh CI run) would otherwise never match. """ - if from_run.dataset_path != to_run.dataset_path: - return False - if sorted(from_run.evaluators) != sorted(to_run.evaluators): - return False return _agent_identity(from_run.target) == _agent_identity(to_run.target) diff --git a/tests/unit/test_regression_insight.py b/tests/unit/test_regression_insight.py index a2c9ada9..b7d8c963 100644 --- a/tests/unit/test_regression_insight.py +++ b/tests/unit/test_regression_insight.py @@ -168,14 +168,25 @@ def test_no_insight_when_agent_identity_differs(): assert regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) is None -def test_no_insight_when_dataset_differs(): +def test_dataset_change_is_attributed_not_blocked(): + """FR-006/build_changed_inputs explicitly attributes a regression to a + dataset change - same_lineage must not require an identical + dataset_path, both because that's a real, attributable cause this + feature exists to surface, and because dataset_path isn't + checkout-stable (local runs persist an absolute, resolved path), so a + committed baseline from a different checkout than the current CI run + would otherwise never match even for the literal same repo-relative + dataset.""" from_run = _run(dataset_path="data/a.jsonl", accuracy=0.91, commit=_commit("a" * 40)) to_run = _run(dataset_path="data/b.jsonl", accuracy=0.79, commit=_commit("b" * 40)) - assert regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) is None + insight = regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) + + assert insight is not None + assert any(c.field == "dataset" for c in insight.changed_inputs) -def test_no_insight_when_evaluators_differ(): +def test_evaluator_set_change_is_attributed_not_blocked(): from_run = _run( evaluators=["CoherenceEvaluator"], accuracy=0.91, commit=_commit("a" * 40) ) @@ -185,7 +196,10 @@ def test_no_insight_when_evaluators_differ(): commit=_commit("b" * 40), ) - assert regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) is None + insight = regression_insight.build_regression_insight(from_run, to_run, metrics=["accuracy"]) + + assert insight is not None + assert any(c.field == "evaluators" for c in insight.changed_inputs) def test_same_lineage_is_version_blind(): @@ -198,6 +212,44 @@ def test_same_lineage_is_version_blind(): assert regression_insight.same_lineage(from_run, to_run) is True +def test_normalize_agent_url_strips_only_the_version_segment(): + url = "https://acct.services.ai.azure.com/api/projects/p/agents/bot/versions/11" + normalized = regression_insight.normalize_agent_url(url) + + assert "/versions/11" not in normalized + assert normalized == "https://acct.services.ai.azure.com/api/projects/p/agents/bot" + + +def test_hosted_agent_same_lineage_across_a_version_bump(): + """A foundry_hosted URL embeds /versions/ - using it as-is for + agent identity would make every version bump start a new lineage, + exactly the change this feature exists to attribute a regression to.""" + base_url = "https://acct.services.ai.azure.com/api/projects/p/agents/bot/versions" + from_run = _run(commit=_commit("a" * 40)) + from_run.target.kind = "foundry_hosted" + from_run.target.url = f"{base_url}/11" + to_run = _run(commit=_commit("b" * 40)) + to_run.target.kind = "foundry_hosted" + to_run.target.url = f"{base_url}/12" + + assert regression_insight.same_lineage(from_run, to_run) is True + + +def test_hosted_agent_different_project_is_different_lineage(): + from_run = _run(commit=_commit("a" * 40)) + from_run.target.kind = "foundry_hosted" + from_run.target.url = ( + "https://acct.services.ai.azure.com/api/projects/project-a/agents/bot/versions/1" + ) + to_run = _run(commit=_commit("b" * 40)) + to_run.target.kind = "foundry_hosted" + to_run.target.url = ( + "https://acct.services.ai.azure.com/api/projects/project-b/agents/bot/versions/1" + ) + + assert regression_insight.same_lineage(from_run, to_run) is False + + def test_no_insight_when_metric_missing_on_either_run(): from_run = _run(commit=_commit("a" * 40)) to_run = _run(commit=_commit("b" * 40)) From d313df444a0a967ec849d9421a90c4686c704ee3 Mon Sep 17 00:00:00 2001 From: exesilvestre Date: Thu, 8 Oct 2026 16:17:47 -0300 Subject: [PATCH 30/30] fix(pipeline): handle boolean threshold criteria; split attribution's --- docs/how-it-works.md | 39 +++++--- src/agentops/agent/checks/regression.py | 33 ++++--- src/agentops/agent/sources/results_history.py | 75 ++++++++++------ src/agentops/pipeline/regression_insight.py | 15 +++- tests/unit/test_agent_checks_regression.py | 88 ++++++++++++++++++- tests/unit/test_pipeline_comparison.py | 21 +++++ 6 files changed, 213 insertions(+), 58 deletions(-) diff --git a/docs/how-it-works.md b/docs/how-it-works.md index c56a56d2..fb51ea1c 100644 --- a/docs/how-it-works.md +++ b/docs/how-it-works.md @@ -725,19 +725,32 @@ Foundry prompt-agent deploy gating), and on a best-effort basis locally via `git rev-parse HEAD`. It's `null` when the workspace isn't a git repository or `git` is unavailable; nothing else about the run changes. -When one or more metrics regress between two comparable runs (same agent -target, dataset, and evaluator set) that both have commit metadata - -whether detected via an explicit `--baseline` comparison or Doctor's -rolling regression check - `report.md` gains a "Regression Insight" section -listing every metric that regressed (not just the worst one) and explaining -what changed between the two commits (system prompt, model, dataset, -evaluators, or thresholds) and suggesting a corrective action. When either -run was published to Foundry (`execution: cloud`, or local `publish: true`), -the section also links out to its Evaluations page. This is deterministic -field-diffing, not an LLM call, and it's purely informational: it never -affects the exit-code/threshold-gating contract. Runs without commit -metadata, or where nothing tracked changed, behave exactly as they did -before this existed. +When one or more metrics regress between two runs of the same agent +identity (deliberately independent of version/model/deployment - a change +there is itself one of the things attributed) that both have commit +metadata, a dataset or evaluator-set change between them is attributable +too, same as a prompt/model change: the system never requires an +identical dataset/evaluator set to explain a regression, only that it's +the same agent. + +For an explicit `--baseline` comparison, this explanation is a +"Regression Insight" section appended to `report.md`, listing every +metric that regressed (not just the worst one) and explaining what +changed between the two commits (system prompt, model, dataset, +evaluators, or thresholds) with a suggested corrective action. When +either run was published to Foundry (`execution: cloud`, or local +`publish: true`), the section also links out to its Evaluations page. +Doctor's rolling regression check surfaces the same explanation +differently: not in `report.md` at all, but in the triggering `Finding`'s +own `recommendation` text (and `evidence["insight"]`) - Doctor's own +rolling-baseline *detection* additionally stays on the stricter, +version-inclusive methodology fingerprint unchanged by any of this (see +`agent.checks.regression`), so a version bump alone doesn't affect +*whether* Doctor flags a regression, only how the already-flagged one +gets explained. This is deterministic field-diffing, not an LLM call, and +it's purely informational: it never affects the exit-code/threshold-gating +contract. Runs without commit metadata, or where nothing tracked changed, +behave exactly as they did before this existed. Every run produced by `agentops eval run` - including `execution: cloud` and `execution: azd` - attempts commit capture. The one case with no commit diff --git a/src/agentops/agent/checks/regression.py b/src/agentops/agent/checks/regression.py index 5ee260fe..2326b772 100644 --- a/src/agentops/agent/checks/regression.py +++ b/src/agentops/agent/checks/regression.py @@ -107,21 +107,26 @@ def run_regression_check( # The immediately preceding comparable run, for causal attribution - # distinct from `baseline_runs` above (detection's rolling drop% mean, - # left untouched). Deliberately keyed on the coarser `lineage_key` - # (dataset, evaluators, agent identity - version/deployment excluded) - # rather than `baseline_runs`/`methodology_fingerprint`: when a - # regression *has* been detected and happens to coincide with a - # version bump, the fingerprint would exclude the pre-bump run from - # `baseline_runs`, so picking the comparison partner from there could - # silently skip over it and attribute the cause to an older, less - # relevant run instead. This only changes which run an *already-fired* - # finding is explained against, never whether one fires. - lineage_key = latest.lineage_key - if lineage_key is None: - lineage_runs = runs[:-1] + # left untouched). Deliberately keyed on `agent_identity_key` alone + # (not `lineage_key`, which also requires the same dataset/evaluators, + # and not `baseline_runs`/`methodology_fingerprint`, which also + # requires the same version): a dataset, evaluator-set, *or* + # version/deployment change must each remain attributable + # (`same_lineage` in `pipeline.regression_insight` enforces the same, + # agent-identity-only rule) - filtering the candidate pool by any of + # those first would make picking the run that actually changed (and + # thus detecting that very change) impossible whenever an intervening + # run differed only in dataset, evaluators, or version. This only + # changes which run an *already-fired* finding is explained against, + # never whether one fires. + agent_identity_key = latest.agent_identity_key + if agent_identity_key is None: + identity_runs = runs[:-1] else: - lineage_runs = [r for r in runs[:-1] if r.lineage_key == lineage_key] - previous_run = lineage_runs[-1] if lineage_runs else None + identity_runs = [ + r for r in runs[:-1] if r.agent_identity_key == agent_identity_key + ] + previous_run = identity_runs[-1] if identity_runs else None latest_result = _load_run_result(latest) previous_result = _load_run_result(previous_run) if previous_run is not None else None diff --git a/src/agentops/agent/sources/results_history.py b/src/agentops/agent/sources/results_history.py index 4818a8fc..23df0157 100644 --- a/src/agentops/agent/sources/results_history.py +++ b/src/agentops/agent/sources/results_history.py @@ -38,10 +38,22 @@ class RunSummary: portal_url: Optional[str] = None methodology_fingerprint: Optional[str] = None # Coarser than `methodology_fingerprint` - same agent identity, dataset, - # and evaluator set, but version/deployment excluded. See - # `_lineage_key`'s docstring for why `agent.checks.regression` needs - # this separate, coarser key for causal attribution. + # and evaluator set, but version/deployment excluded. Used for + # Cockpit-style display grouping (same grouping concept as + # `cockpit._version_lineage_key`) - *not* for picking `previous_run` in + # `agent.checks.regression` (see `agent_identity_key` below for that). lineage_key: Optional[str] = None + # Narrower still than `lineage_key`: agent identity alone, dataset and + # evaluators excluded. `agent.checks.regression` needs this - not + # `lineage_key` - to pick "the immediately preceding run with the same + # agent" for causal attribution, consistent with + # `regression_insight.same_lineage`'s own, equally agent-identity-only + # rule: a dataset/evaluator-set change is itself one of the things this + # feature must be able to attribute a regression to, so filtering the + # candidate pool by them first would make picking the *actual* + # immediately-preceding run (and thus detecting that very change) + # impossible whenever an intervening run used a different dataset. + agent_identity_key: Optional[str] = None @dataclass @@ -146,39 +158,44 @@ def _summarize(path: Path) -> Optional[RunSummary]: item_evaluations=item_evaluations, methodology_fingerprint=_methodology_fingerprint(data), lineage_key=_lineage_key(data), + agent_identity_key=_agent_identity_key(data), ) +def _agent_identity_key(data: Dict[str, Any]) -> Optional[str]: + """The run's agent identity alone (see + ``regression_insight.agent_identity_from_fields`` - the single source + of truth for this, shared with Cockpit (``cockpit._version_lineage_key``) + and ``build_regression_insight``'s own comparability check), with no + dataset/evaluator information folded in. Feeds ``RunSummary + .agent_identity_key`` - see that field's docstring for why + ``agent.checks.regression`` needs exactly this, narrower than + ``_lineage_key`` below, to pick the comparison partner for causal + attribution. + """ + raw_target = data.get("target") + target: Dict[str, Any] = raw_target if isinstance(raw_target, dict) else {} + identity = agent_identity_from_fields( + name=target.get("name"), + url=target.get("url"), + kind=target.get("kind"), + raw=target.get("raw"), + ) or (data.get("config") or {}).get("agent") + return str(identity) if identity else None + + def _lineage_key(data: Dict[str, Any]) -> Optional[str]: """Derive a stable hash of (agent identity, dataset, evaluators) - deliberately excluding the agent's version/deployment, unlike - ``_methodology_fingerprint`` below. - - ``agent.checks.regression`` needs this coarser key to pick "the - immediately preceding comparable run" for causal regression - attribution. Using ``_methodology_fingerprint`` for that (as it - correctly does for its own drop%-mean baseline, where mixing - methodologies would be spurious) would make the feature's own - canonical scenario - attributing a regression to a version bump, - e.g. "run v3 -> v4" - unreachable: a version bump changes the - fingerprint, so the prior (different-version) run would always be - filtered out before it could be picked as the comparison partner. + ``_methodology_fingerprint`` below. This is the *display* grouping key + (same concept as ``cockpit._version_lineage_key``) - for picking the + comparison partner for causal regression attribution, + ``agent.checks.regression`` uses the narrower ``_agent_identity_key`` + above instead (see its docstring and ``RunSummary.agent_identity_key`` + for why: a dataset/evaluator change must itself remain attributable, + which filtering the candidate pool by them first would prevent). """ - raw_target = data.get("target") - target: Dict[str, Any] = raw_target if isinstance(raw_target, dict) else {} - # See `regression_insight.agent_identity_from_fields`'s docstring - the - # single source of truth for this, shared with Cockpit - # (cockpit._version_lineage_key) and build_regression_insight's own - # comparability check, so the three can't drift apart again. - agent_identity = ( - agent_identity_from_fields( - name=target.get("name"), - url=target.get("url"), - kind=target.get("kind"), - raw=target.get("raw"), - ) - or (data.get("config") or {}).get("agent") - ) + agent_identity = _agent_identity_key(data) dataset_path = data.get("dataset_path") or (data.get("config") or {}).get( "dataset" ) diff --git a/src/agentops/pipeline/regression_insight.py b/src/agentops/pipeline/regression_insight.py index 680ded9a..d65499df 100644 --- a/src/agentops/pipeline/regression_insight.py +++ b/src/agentops/pipeline/regression_insight.py @@ -86,13 +86,26 @@ def metric_threshold_criteria( def _is_lower_is_better(metric: str, *, criteria: Optional[str] = None) -> bool: """Whether a lower value is the better outcome for ``metric``. - ``criteria`` (a recorded threshold operator like ``"<="``, from + ``criteria`` (a recorded threshold operator, from ``metric_threshold_criteria``) wins when given - it reflects what this specific metric was actually configured to mean, unlike ``LOWER_IS_BETTER_METRICS``, which is only a fallback guess for the one built-in metric known to commonly be lower-is-better. + + A boolean threshold (``"false"``/``"true"``, for a metric whose values + are 0/1) is treated the same as ``<=``/``>=``: passing means 0 + (``"false"``), so lower is better, same as it would be for a + continuous metric thresholded that way. An equality threshold + (``"=="``) has no well-defined "lower is better" at all - what's + "closer to passing" depends on the recorded target value, not raw + magnitude - so it's deliberately left unhandled here and falls through + to the same default as no criteria at all, rather than guessing. """ if criteria is not None: + if criteria == "false": + return True + if criteria == "true": + return False if criteria.startswith("<"): return True if criteria.startswith(">"): diff --git a/tests/unit/test_agent_checks_regression.py b/tests/unit/test_agent_checks_regression.py index 41f3cfc6..2f9fcac7 100644 --- a/tests/unit/test_agent_checks_regression.py +++ b/tests/unit/test_agent_checks_regression.py @@ -18,6 +18,7 @@ def _run( offset_days: int = 0, fingerprint: str | None = None, lineage_key: str | None = None, + agent_identity_key: str | None = None, raw_path: Path | None = None, source: str = "local", ) -> RunSummary: @@ -31,6 +32,7 @@ def _run( raw_path=raw_path or Path("dummy"), methodology_fingerprint=fingerprint, lineage_key=lineage_key, + agent_identity_key=agent_identity_key, source=source, ) @@ -42,6 +44,7 @@ def _write_result_json( version: str, deployment: str, commit_sha: str, + dataset_path: str = "data/smoke.jsonl", ) -> None: payload = { "version": 1, @@ -55,7 +58,7 @@ def _write_result_json( "version": version, "deployment": deployment, }, - "dataset_path": "data/smoke.jsonl", + "dataset_path": dataset_path, "evaluators": ["CoherenceEvaluator"], "rows": [], "aggregate_metrics": {"accuracy": accuracy}, @@ -374,6 +377,7 @@ def test_attribution_uses_the_true_immediately_preceding_run_across_a_version_ch offset_days=-2, fingerprint="V4", lineage_key="L", + agent_identity_key="L", raw_path=v4_path, ), _run( @@ -382,6 +386,7 @@ def test_attribution_uses_the_true_immediately_preceding_run_across_a_version_ch offset_days=-1, fingerprint="V5", lineage_key="L", + agent_identity_key="L", raw_path=v5_path, ), _run( @@ -390,6 +395,7 @@ def test_attribution_uses_the_true_immediately_preceding_run_across_a_version_ch offset_days=0, fingerprint="V4", lineage_key="L", + agent_identity_key="L", raw_path=latest_path, ), ] @@ -466,6 +472,7 @@ def test_no_insight_when_the_selected_pair_did_not_itself_regress(tmp_path) -> N offset_days=-2, fingerprint="V4", lineage_key="L", + agent_identity_key="L", raw_path=v4_path, ), _run( @@ -474,6 +481,7 @@ def test_no_insight_when_the_selected_pair_did_not_itself_regress(tmp_path) -> N offset_days=-1, fingerprint="V5", lineage_key="L", + agent_identity_key="L", raw_path=v5_path, ), _run( @@ -482,6 +490,7 @@ def test_no_insight_when_the_selected_pair_did_not_itself_regress(tmp_path) -> N offset_days=0, fingerprint="V4", lineage_key="L", + agent_identity_key="L", raw_path=latest_path, ), ] @@ -499,3 +508,80 @@ def test_no_insight_when_the_selected_pair_did_not_itself_regress(tmp_path) -> N "baseline instead" ) assert "didn't show this drop" in finding.recommendation + + +def test_attribution_picks_the_immediately_preceding_run_across_a_dataset_change( + tmp_path, +) -> None: + """Older runs used dataset B, one intervening run used dataset A, then + the latest (regressed) run is back on dataset B. Filtering candidates + by `lineage_key` (which also requires the same dataset) would skip the + dataset-A run and miss that the dataset changed in between - + `agent_identity_key` (agent identity only) must pick it instead. + """ + b1_path = tmp_path / "b1" / "results.json" + a_path = tmp_path / "a" / "results.json" + latest_path = tmp_path / "latest" / "results.json" + _write_result_json( + b1_path, + accuracy=0.90, + version="4", + deployment="gpt-4o", + commit_sha="1" * 40, + dataset_path="data/dataset-b.jsonl", + ) + _write_result_json( + a_path, + accuracy=0.92, + version="4", + deployment="gpt-4o", + commit_sha="2" * 40, + dataset_path="data/dataset-a.jsonl", + ) + _write_result_json( + latest_path, + accuracy=0.70, + version="4", + deployment="gpt-4o", + commit_sha="3" * 40, + dataset_path="data/dataset-b.jsonl", + ) + + history = ResultsHistory( + runs=[ + _run( + {"accuracy": 0.90}, + run_id="b1", + offset_days=-2, + fingerprint="B", + agent_identity_key="agent", + raw_path=b1_path, + ), + _run( + {"accuracy": 0.92}, + run_id="a", + offset_days=-1, + fingerprint="A", + agent_identity_key="agent", + raw_path=a_path, + ), + _run( + {"accuracy": 0.70}, + run_id="latest", + offset_days=0, + fingerprint="B", + agent_identity_key="agent", + raw_path=latest_path, + ), + ] + ) + config = RegressionCheckConfig(metrics=["accuracy"], threshold_drop=0.10, min_runs=2) + + findings = run_regression_check(history, config) + + assert len(findings) == 1 + insight = findings[0].evidence["insight"] + # The dataset-A run's commit, not the older dataset-B run's. + assert insight["from_commit"]["sha"] == "2" * 40 + fields = {c["field"] for c in insight["changed_inputs"]} + assert "dataset" in fields diff --git a/tests/unit/test_pipeline_comparison.py b/tests/unit/test_pipeline_comparison.py index 8337fd45..62ff20b9 100644 --- a/tests/unit/test_pipeline_comparison.py +++ b/tests/unit/test_pipeline_comparison.py @@ -187,6 +187,27 @@ def test_custom_lower_is_better_metric_drop_is_improved_not_regressed(): assert info.insight is None +def test_boolean_false_threshold_metric_drop_is_improved(): + """A `thresholds: { has_error: false }` metric (values 0/1) passes + when it's 0 - moving from 1 to 0 is an improvement, same direction as + a `<=` threshold, not the higher-is-better default.""" + baseline = _with_threshold( + _run(accuracy=0.91, commit=_commit("a" * 40)), metric="has_error", criteria="false" + ) + baseline.aggregate_metrics["has_error"] = 1.0 + current = _with_threshold( + _run(accuracy=0.91, commit=_commit("b" * 40)), metric="has_error", criteria="false" + ) + current.aggregate_metrics["has_error"] = 0.0 + + info = comparison.build_comparison( + current=current, baseline=baseline, baseline_path=Path(".agentops/baseline/results.json") + ) + + has_error_metric = next(m for m in info.metrics if m.metric == "has_error") + assert has_error_metric.direction == "improved" + + def test_insight_carries_baseline_report_url_from_sidecar_file(tmp_path: Path): """``baseline`` is a prior, fully-published run - a sidecar ``cloud_evaluation.json`` next to its ``results.json`` (as a completed