diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index 7c79968..9c978bd 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -3,8 +3,18 @@ on:
push:
branches: [main]
pull_request:
+permissions:
+ contents: read
jobs:
+ msrv:
+ runs-on: ubuntu-latest
+ timeout-minutes: 10
+ steps:
+ - uses: actions/checkout@v7
+ - uses: dtolnay/rust-toolchain@1.78.0
+ - run: cargo build --release --locked --workspace
test:
+ timeout-minutes: 30
strategy:
fail-fast: false
matrix:
@@ -13,6 +23,9 @@ jobs:
steps:
- uses: actions/checkout@v7
- uses: dtolnay/rust-toolchain@stable
+ - uses: actions/setup-node@v6
+ with:
+ node-version: '24'
- name: build
run: cargo build --release --workspace
- name: test
@@ -26,13 +39,61 @@ jobs:
run: cargo run --release -p probbit-cli -- stats --pretty
- name: agent_router example
run: cargo run --release --example agent_router
+ - name: Python and MCP integration contracts
+ shell: bash
+ run: |
+ B="$PWD/target/release/probbit"
+ if [ "$RUNNER_OS" = Windows ]; then B="$(cygpath -w "$B.exe")"; fi
+ export PROBBIT_BIN="$B"
+ python -m unittest discover -s python -p 'test_*.py'
+ node examples/node/decide.mjs
+ - name: npm platform and process contracts
+ run: node --test npm/test-wrapper.cjs
+ - name: clean-prefix install and npm package contracts
+ shell: bash
+ run: |
+ set -euo pipefail
+ TARGET=$(rustc -vV | sed -n 's/^host: //p')
+ TAG=v$(node -p "require('./npm/package.json').version")
+ BIN=target/release/probbit
+ [ "$RUNNER_OS" = Windows ] && BIN="$BIN.exe"
+ if [ "$RUNNER_OS" != Windows ]; then
+ python3 scripts/tests/test_install_contract.py
+ sh scripts/bench/test_install_sh.sh "$BIN" "$TARGET" "$TAG"
+ else
+ sh scripts/bench/package_like_release.sh "$BIN" "$TARGET" "$TAG" "target/install-fixture/good/$TAG"
+ cp -R target/install-fixture/good target/install-fixture/bad
+ printf '%064d %s\n' 0 "probbit-$TAG-$TARGET.zip" > "target/install-fixture/bad/$TAG/probbit-$TAG-$TARGET.zip.sha256"
+ for PS in pwsh powershell; do
+ "$PS" -NoProfile -ExecutionPolicy Bypass -File scripts/bench/test_install_ps1.ps1 -Srv target/install-fixture -Tag "$TAG"
+ done
+ fi
+ sh scripts/bench/test_npm.sh "$BIN" "$TARGET" "$TAG"
+ - name: npm 12 clean-prefix contracts
+ if: runner.os == 'Linux'
+ shell: bash
+ run: |
+ npm install -g npm@12
+ sh scripts/bench/test_npm.sh target/release/probbit x86_64-unknown-linux-gnu
+ npm-node18:
+ runs-on: ubuntu-latest
+ steps:
+ - uses: actions/checkout@v7
+ - uses: actions/setup-node@v6
+ with:
+ node-version: '18'
+ - run: node --test npm/test-wrapper.cjs
wasm:
+ timeout-minutes: 15
# probbit-wasm, the browser build (wasm32-unknown-unknown, no threads): built as playground/build.sh builds it, then loaded by
# Node with the playground's own loader and run on the 300-task demo (--sweeps 3200) and the evaluate example; check.mjs
# exits 1 unless both answer with 0 violations. The no-thread = 4-thread equality runs natively in `test` (probbit-wasm tests).
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
+ - uses: actions/setup-node@v6
+ with:
+ node-version: '24'
- uses: dtolnay/rust-toolchain@stable
with:
targets: wasm32-unknown-unknown
@@ -40,6 +101,22 @@ jobs:
run: sh playground/build.sh
- name: run under Node
run: node --version && node playground/check.mjs playground/probbit.wasm
+ - name: WASM integer boundaries and native parity
+ run: |
+ cargo build --release -p probbit-cli
+ node probbit-wasm/tests/boundaries.mjs playground/probbit.wasm target/release/probbit
+ - name: real browser interaction and narrow layout
+ run: node playground/check-browser.mjs target/browser-check
+ - name: real monitor playback and live append in the browser
+ env:
+ PROBBIT_BIN: ${{ github.workspace }}/target/release/probbit
+ PROBBIT_BROWSER_OUT: ${{ github.workspace }}/target/browser-check/monitor
+ run: node probbit-cli/tests/monitor_browser.mjs
+ - uses: actions/upload-artifact@v7
+ if: always()
+ with:
+ name: browser-check
+ path: target/browser-check/
release-targets:
# The release archives' targets that `test` does not build (release.yml runs only on tags): built here, so a tag is
# never the first build of a target. Static musl (x86_64 runs here, aarch64 is cross-linked), macOS x86_64 under
@@ -50,6 +127,7 @@ jobs:
include:
- { os: ubuntu-latest, target: x86_64-unknown-linux-musl, run: true }
- { os: ubuntu-latest, target: aarch64-unknown-linux-musl, linker: aarch64-linux-gnu-gcc }
+ - { os: ubuntu-24.04-arm, target: aarch64-unknown-linux-musl, linker: aarch64-linux-gnu-gcc, run: true }
- { os: macos-latest, target: x86_64-apple-darwin, run: true, rosetta: true }
- { os: windows-latest, target: x86_64-pc-windows-msvc, run: true, rustflags: "-C target-feature=+crt-static" }
runs-on: ${{ matrix.os }}
diff --git a/.gitignore b/.gitignore
index 8b6b2f7..992daf3 100644
--- a/.gitignore
+++ b/.gitignore
@@ -10,3 +10,28 @@ __pycache__/
playground/probbit.wasm
playground/probbit-wasm.js
playground/puzzle-personas.js
+
+# Local credentials, agent context and generated release artifacts are not product sources.
+.env
+.env.*
+*.credentials*
+*credentials.json
+cookies.txt
+login_response*
+.secrets/
+.agents/
+.ouroboros/
+AGENTS.md
+SOUL.md
+USER.md
+MEMORY.md
+IDENTITY.md
+memory/
+identity/
+diary/
+state/
+audits/
+data/
+node_modules/
+dist/
+*.tgz
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 242da1e..216e2e5 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -3,6 +3,39 @@
All notable changes to this project are documented here. The format follows
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
+## Unreleased
+
+### Fixed
+- Reject invalid JSON number grammar and raw control characters, platform-sized integer overflow, overflowing linear-cap
+ totals, and invalid pair-table dimensions before allocation. WASM no longer wraps large precedence gaps, counts or weights.
+- Validate supplied anneal/polish starting assignments even when the work budget is zero.
+- Validate stored strand hashes and lifecycle transitions before continuation or control appends; a checkpoint cannot hide
+ retirement. Full inference verification and external anchoring remain separate responsibilities.
+- Python rejects an explicitly missing binary instead of falling back, and rejects invalid false/array input values rather
+ than silently treating them as empty objects. MCP and Python preserve command-specific error/lifecycle contracts.
+- Persona `--pretty` consistently formats single-document stdout without changing canonical state files. JSONL replay and
+ text explain reject the flag with guidance. Source-restricted fuzz exports require an explicitly labelled synthetic source
+ to produce runnable replay commands; production source checks are not bypassed.
+- npm forwards each termination signal unchanged, bounds download waits and selects both x86_64 and ARM64 musl assets.
+ Shell installation defaults to the user's `~/.local/bin`. Installers validate candidate binaries before replacing a
+ working installation on the shell/PowerShell paths; PowerShell stages replacement on the destination filesystem.
+ Release/target selectors reject paths.
+- Monitor labels scripted demos, loop restarts, lifecycle and checkpoint status; preserves original `why` text, explains
+ display terms and supports narrow screens. Browser playground and puzzle tables no longer force horizontal page overflow.
+- Keep the lockfile readable by the declared Rust 1.78 minimum; CI builds that compiler with `--locked`.
+
+### Added
+- Decision documents with a candidate `plan` gain `plan_status` and `released_plan`. The old full candidate is preserved for
+ compatibility; a diagnostic/refused plan is not permission to act. Partial projections may not be independently feasible.
+- `probbit monitor --demo drives` shows eight replayable synthetic goal events. The default tutor demo is unchanged.
+- `examples/agent-harness`: a local dispatch gate with a recorded incident, history-dependent retry regression, strict
+ replay and a scoped repaired-rule proof. No model, credentials or external tool action required.
+- Independent enumeration/router oracles, actual WASM boundary tests, real Chromium interaction/layout checks, Node process
+ contracts and cross-platform clean-prefix install tests (including Windows PowerShell 5.1/7 and npm 12).
+
+These changes are a source candidate, not an update to the published 0.8.0 artifacts. Probability gate thresholds and frozen
+benchmark corpora are unchanged. Passing finite tests is not exhaustive validation of every possible program or application.
+
## 0.8.0 - 2026-10-09
### Added
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
index c7374cb..360de3f 100644
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -7,7 +7,7 @@ rules that keep its claims honest.
```
cargo build --release --workspace # no network needed: there are no external crates
-cargo test --release --workspace # 127 tests: 3 core, 48 CLI, 11 stress (+1 ignored), 18 acceptance, 47 probbit-ir
+cargo test --release --workspace # unit, integration, regression and exact-oracle tests
cargo run --release -p probbit-cli -- demo --tasks 12 | cargo run --release -p probbit-cli -- decide --pretty
```
@@ -19,6 +19,11 @@ loaded machine; re-run it alone before reading anything into a failure.
The default build is portable (no CPU pin). `RUSTFLAGS="-C target-cpu=native"` gives the last
bit of speed on your own machine; that binary may not run elsewhere.
+Release-contract checks also run on every pull request: the Python and MCP suites, npm platform/signal tests,
+and clean-prefix shell, PowerShell 5.1/7 and npm installs using locally packaged candidate binaries.
+`node --test npm/test-wrapper.cjs` needs no npm dependencies. Set `PROBBIT_BIN` to the release executable and run
+`python3 -m unittest discover -s python -p 'test_*.py'` for the integration contracts.
+
## Ground rules
- **No external crates on the shipped path.** `cargo build` must keep working offline.
diff --git a/Cargo.lock b/Cargo.lock
index 6a5f8ce..ded6842 100644
--- a/Cargo.lock
+++ b/Cargo.lock
@@ -1,6 +1,6 @@
# This file is automatically @generated by Cargo.
# It is not intended for manual editing.
-version = 4
+version = 3
[[package]]
name = "probbit-cli"
diff --git a/PORTABILITY.md b/PORTABILITY.md
index 376d172..3b31faf 100644
--- a/PORTABILITY.md
+++ b/PORTABILITY.md
@@ -122,8 +122,9 @@ No release exists yet, so every installer was tested against a local server that
| npm wrapper (`npm/`) | `ubuntu-latest`, `windows-latest`, local M4 | `npm pack`, `npm install -g` (scratch prefix), postinstall fetch and SHA-256 check, `which probbit`, exit codes 0 / 1 / 2 / 3 passed through, `npm uninstall -g`; an `--ignore-scripts` install fetches on first run; `PROBBIT_BINARY`; a wrong `.sha256` fails the install |
| `docs/agents.md` recipes | shell, Python (`python/test_probbit.py`), Node (`examples/node/decide.mjs`): Linux, macOS arm64 and x86_64, Windows (Git Bash); PowerShell 7.6: Windows | each recipe as written, plus one call per exit code |
-Install locations: `install.sh` writes `/usr/local/bin` when it can (every runner image above: their user can write it)
-and `~/.local/bin` otherwise (the local M4, where `/usr/local/bin` belongs to root), and never uses sudo; `install.ps1`
+Historical install locations in the measurements above: `install.sh` wrote `/usr/local/bin` when writable,
+and `~/.local/bin` otherwise. The current source defaults to `~/.local/bin` on every shell host; use
+`PROBBIT_INSTALL_DIR` for an explicit system-wide destination. It never uses sudo. `install.ps1`
writes `$HOME\.local\bin`, adds it to the session's PATH, and to the user PATH only with `-AddToPath`. Names: `probbit` is
free on npm and `probbit`, `probbit-core`, `probbit-ir`, `probbit-decide` and `probbit-cli` are free on crates.io (checked 2026-10-01
22:31 PDT); nothing was published.
@@ -139,7 +140,8 @@ free on npm and `probbit`, `probbit-core`, `probbit-ir`, `probbit-decide` and `p
2. **The Linux release binary needs glibc 2.34.** `release.yml` builds `x86_64-unknown-linux-gnu` on `ubuntu-latest`.
The static musl build above runs on any x86_64 Linux and was within 4% (run 2) and 7% (run 1) of the glibc build on
the same VM, with 2.5-3.2 MiB less peak memory (BENCHMARK-MATRIX.md). Shipping it, plus `aarch64-unknown-linux-musl`,
- would also give `install.sh` something for Alpine and arm64 Linux, which it refuses today.
+ now provides the released binaries for Alpine and arm64 Linux. The current shell/npm installers select these
+ targets automatically; the earlier refusal described in these historical measurements is superseded.
3. **The Windows binary needs `VCRUNTIME140.dll`.** Linking the CRT statically (`-C target-feature=+crt-static` for the
msvc release build) would remove that; not built or measured here.
4. **Two wall-clock test bounds failed on shared runners** in run 1: `run_deadline_ms_bounds_the_whole_call` on
diff --git a/README.md b/README.md
index 5f92ff7..9d52e8e 100644
--- a/README.md
+++ b/README.md
@@ -118,6 +118,11 @@ were verified from clean machines on macOS (Apple silicon) and Linux (x86_64).
npm may print an `allow-scripts` warning: the package's only install script downloads the prebuilt binary for your platform
and checks its SHA-256. If npm blocks the script, the binary is fetched the first time you run `probbit` instead.
+The shell installer defaults to `~/.local/bin`, without replacing a system-wide installation. Set
+`PROBBIT_INSTALL_DIR` to choose another directory. Both the shell and npm installers select static musl binaries
+for Linux ARM64 and for x86_64 musl systems such as Alpine. The shell and PowerShell installers check that a
+downloaded candidate starts successfully before replacing an existing binary; failed checksums leave it unchanged.
+
From source, with Rust 1.78 or later: `cargo install --git https://github.com/BitmapAsset/probbit probbit-cli`, or clone and
`cargo build --release -p probbit-cli` (the binary lands in `target/release/probbit`). Nothing is downloaded after the clone:
there are no external crates.
@@ -128,7 +133,7 @@ At a terminal, `install.sh` ends with probbit's own hero screen. Platform notes
```
git clone https://github.com/BitmapAsset/probbit && cd probbit
-cargo test --release --workspace # 127 tests (3 core, 48 CLI, 11 stress, 18 acceptance, 47 probbit-ir; 1 ignored), ~25 s once built
+cargo test --release --workspace # unit, integration, regression and exact-oracle tests
cargo run --release -p probbit-cli -- demo --tasks 12 | cargo run --release -p probbit-cli -- decide --pretty
cargo run --release --example agent_router # the full narrated demo, ~3 s
```
@@ -437,11 +442,11 @@ to be measured per model (no such measurement has been made here).
### Test a character
-
The fuzz finds the flaw: the shortest event script that pushes an individual out of character, shrunk and replayable.
+
The fuzz finds the flaw: an event script that pushes an individual out of character, shrunk and replayable.
A character property is a rule in habit syntax the stance must never break. `fuzz` searches event scripts for each
-individual's shortest counterexample; `prove` says `held by construction`, `proved for every event sequence` or `unknown`. The
+individual's short counterexample (not a guaranteed global minimum); `prove` says `held by construction`, `proved for every event sequence` or `unknown`. The
tutor as it shipped in 0.5.0 is kept as a fixture, so this runs from a clone with no keys and no model:
```sh
@@ -529,6 +534,7 @@ anywhere ([docs/persona.md](docs/persona.md) §5.8). No strand yet? The tutor's
```sh
probbit monitor --demo --open
+probbit monitor --demo drives --open # source checkout: synthetic goals/drive demo, not yet in the 0.8.0 release
```

diff --git a/USE-CASES.md b/USE-CASES.md
index be4e75b..a4d7736 100644
--- a/USE-CASES.md
+++ b/USE-CASES.md
@@ -2,9 +2,12 @@
**One processor, many kinds of decision.** If you can write a decision as choices, scores and rules that must never break,
it runs on the same p-bit engine: tasks to agents, steps to tools, jobs to time slots, items under a budget, cells of a
-puzzle, the temperament of an agent. Every shape below gets the same three things back: a plan that keeps every rule by
-construction, odds for every part (exact whenever the structure lets probbit count them), and a verdict that refuses when its
+puzzle, the temperament of an agent. A feasible answer supplies a rule-abiding candidate plan,
+odds (exact whenever the structure lets probbit count them), and a verdict that refuses when its
diagnostics fail. And every shape gets the same honesty: where a solver, an annealer or plain backtracking wins, the table says so.
+An infeasible or resource-limited problem need not produce a plan. A retained candidate on refusal is diagnostic only;
+check `plan_status`, `released_plan` and the escalation set before host actions. This is a finite-domain probabilistic
+decision processor, not a replacement for arbitrary CPU computation or a guarantee that any application will fit its limits.
Every "gives" cell below points to a measured table in `BENCHMARKS.md` (§ numbers) or a separate run; every "do not promise"
cell points to where probbit lost or was not tested. All measurements are on one Apple M4 Mac mini with synthetic or generated
@@ -30,7 +33,7 @@ Do not use it when:
| problem shape | how it maps to the probbit instruction set (`probbit-ir`) | what probbit gives (measured) | do not promise | who might use it *(inferred)* |
|---|---|---|---|---|
-| **Assign items to resources under rules** (tasks to agents/models/people, tickets to agents, jobs to machines) with quotas and a same-group preference | one categorical variable per item over its allowed resources; unary log-weights = your scores; at-most-`cap` per resource; a Potts pair per same-group pair (`probbit decide` lowers to this) | **0 rule violations by construction** (300-task demo: per-task argmax breaks 144 rules incl. 60 PII tasks sent to cloud models; probbit 0); **per-item odds you can check**: 1,731 released tasks on queues with an exact answer, 0 wrong (§2.1, current gate, re-run on 0.2.0; 1,916 on the 0.1.0 gate); **refusal/escalation** by budget (§2.3, demo 2 below); **exact odds in 0.17-28 ms** up to ~24 tasks (§2.1, §2.3); what-if clamps, including clamps that fill a worker (acceptance test `forced_caps_do_not_freeze_the_gate`: 3 queues whose clamps fill a worker + 1 forced queue each pass the gate within 0.05 TV of exact, green on the current build; a 0.1.0 record of 10/10 passed, 0 false, was not re-run) | the single best plan (ILP wins, §2.2); full coverage on saturated or strong-affinity queues (51% of tasks released on the oracle queues on the current gate, 58% on 0.1.0, §2.1); speed (the sampled path spends its ~0.3 s budget by design, §2.3) | AI platform teams routing LLM tasks; support operations; internal-tools developers |
+| **Assign items to resources under rules** (tasks to agents/models/people, tickets to agents, jobs to machines) with quotas and a same-group preference | one categorical variable per item over its allowed resources; unary log-weights = your scores; at-most-`cap` per resource; a Potts pair per same-group pair (`probbit decide` lowers to this) | **0 rule violations by construction** (300-task `agent_router` example: per-task argmax breaks 144 rules incl. 60 PII tasks sent to cloud models; probbit 0); **per-item odds you can check**: 1,731 released tasks on queues with an exact answer, 0 wrong (§2.1, current gate, re-run on 0.2.0; 1,916 on the 0.1.0 gate); **refusal/escalation** by budget (§2.3, demo 2 below); **exact odds in 0.17-28 ms** up to ~24 tasks (§2.1, §2.3); what-if clamps, including clamps that fill a worker (acceptance test `forced_caps_do_not_freeze_the_gate`: 3 queues whose clamps fill a worker + 1 forced queue each pass the gate within 0.05 TV of exact, green on the current build; a 0.1.0 record of 10/10 passed, 0 false, was not re-run) | the single best plan (ILP wins, §2.2); full coverage on saturated or strong-affinity queues (51% of tasks released on the oracle queues on the current gate, 58% on 0.1.0, §2.1); speed (the sampled path spends its ~0.3 s budget by design, §2.3) | AI platform teams routing LLM tasks; support operations; internal-tools developers |
| **Binary pairwise models** (Ising, max-cut, binary Markov random fields) | one two-valued variable per spin; `table` pairs = couplings; unary = fields | **per-spin odds whose gate verdict was checked against brute force**: 144 runs, 61 passed the gate, 0 false whole answers, 929 released spins, 0 wrong (§3, re-run on 0.2.0; the 0.1.0 gate: 72 passed, 1,346 released, 0 wrong); exact odds and log Z by the exact tiers on small models; at equal time, cuts within 0.06-0.7% of simulated annealing in anneal-only mode and 0.17-1.75% short when sampling (the mode that gives odds, §3) | better cuts than simulated annealing (it won by 4-7 edges, §3); log Z from the sampler (not implemented); correctness beyond 16 spins (the brute-force oracle stops there) | researchers and teachers of p-bit / Ising computing; people prototyping before renting annealer time |
| **Constraint satisfaction with odds** (graph colouring, one-hot puzzles, slot assignment) | a categorical variable per node over colours/digits; negative Potts or at-most-1 caps for "differ"; clamps for givens | **exact per-variable odds and solution counts** from the exact tier (sudoku odds equal a backtracking count on 20 puzzles; colouring 0.00-5.70 ms median, §4); the sampler released 525 vertices, 0 outside tolerance (§4; 468 in an earlier build) | speed against backtracking (15-102x slower per puzzle, §4); sampler coverage on frozen or few-solution instances (sudoku: refused 20/20; 3-colouring: 9 of 14 refused on the default set in the current build, 59 of 82 over 6 sets, §4) | CSP and puzzle tooling, teaching, frequency/register-assignment prototypes *(unmeasured on real instances)* |
| **Scheduling with capacity per time slot** (unit-time jobs, release/deadline windows, precedence) | a variable per job over slots; allowed = window; one cap per slot; precedence i -> j as pair caps "not (i at a and j at b)" for a >= b (O(T^2) caps per edge) | **exact odds on small instances** and a gate that released 428 jobs on 39 oracle instances, 0 outside tolerance; at 200-500 jobs, anneal from a greedy plan beat restarted greedy + repair at equal 100 ms on 4/5 seeds each (§4.3) | optimal schedules at scale (an ILP solver, HiGHS, proves the 200-job optimum in 0.68-0.95 s; probbit's plan is 3.3-4.3 nats short, §4.3); checked odds at scale (at 200 jobs the sampler needs 0.7 s just to start, refuses at 1 s, releases 153-171/200 at 5 s on 3 instances in an earlier build (re-run on a later build, N = 3: 156-173 on two, the third refused) (earlier build, start inside the budget; 2 runs each) with no oracle to check them); multi-period durations (unit jobs only) | planners prototyping small shift/slot problems *(inferred)* |
@@ -61,12 +64,15 @@ Do not use it when:
- **CLI + JSON** on stdin/stdout, exit codes 0 plan / 1 infeasible / 2 bad input / 3 refused. Every `bench/` script drives it.
Process overhead is ~2-5 ms (§2.3: 9.0 ms wall vs 7.1 ms inside probbit at 12 tasks, 303.1 vs 299.0 at 300; 0.2.0 re-run 10.1 vs 8.0
and 314.3 vs 309.4).
-- **Python**: `python/probbit.py` (0.2.0), a stdlib-only subprocess wrapper: `run` / `exact` / `sample` / `decide`, typed errors,
- `deadline_ms`; 9 tests run inside `cargo test`; three examples in `python/examples/`. No native bindings (one process per call).
+- **Python**: `python/probbit.py`, a stdlib-only subprocess wrapper for decisions, evaluation, personas and live events;
+ typed errors and deadlines. CI runs unittest discovery over `python/`. No native bindings (one process per call).
- **Schema**: `docs/probbit-ir.schema.json` (JSON Schema of the probbit-ir wire format; a test keeps it equal to the parser);
`probbit --help` lists every flag with its default.
- **Rust**: the `probbit-decide` crate (router front-end) and `probbit-ir` (general programs, JSON v1 in `docs/probbit-ir-json.md`).
-- **MCP server**: not built (README roadmap).
+- **MCP server**: `probbit mcp` serves the ten documented tools over stdio; see [docs/agents.md](docs/agents.md).
+- **Host-enforced incident regression**: [examples/agent-harness](examples/agent-harness/README.md) records a local tool
+ failure, reproduces history-dependent retries, gates dispatch and checks a repaired rule. A bounded example, not a
+ production authorization system.
- **Running beside other work**: `--threads`, `--cpu-limit`, `--priority low`, `--mem-limit-mb`, `--progress`, `probbit stats` (§5).
Reproducible runs: fixed `--sweeps` plus `--polish-sweeps`. The defaults are wall-clock (sampling and polish), so the odds,
the verdict, the released set and the plan can all vary run to run and machine to machine.
diff --git a/docs/agents.md b/docs/agents.md
index 4a7b47c..9d31278 100644
--- a/docs/agents.md
+++ b/docs/agents.md
@@ -11,7 +11,16 @@ an agent harness with a shell tool. No server, no bindings, no network. The docu
| 0 | an answer: verdict `exact`, `diagnostics_passed` or `partial` | act on `released`; escalate `escalated` (empty unless `partial`) |
| 1 | `infeasible`: no plan satisfies the rules (a proof) | relax a rule or a cap; it is an answer, not a crash |
| 2 | bad input: one `{"error": {"code", "path", "message"}}` object on stdout; a bad flag prints one line on stderr instead | fix the document or the flag |
-| 3 | `refused` / `declined` (the gate or a cap said no), or `{"error": {"code": "numeric"}}` | escalate the whole decision; the best-effort plan is still in the output |
+| 3 | `refused` / `declined` (the gate or a cap said no), or `{"error": {"code": "numeric"}}` | escalate the whole decision; any retained candidate plan is diagnostic only |
+
+Decision documents that carry a `plan` also carry `plan_status` (`released`, `partial`, or `diagnostic`)
+and `released_plan`, which projects the candidate onto the released IDs. On refusal this projection is empty;
+some early failures have no plan at all. The legacy `plan` remains for inspection, not unconditional execution.
+A partial projection is **not** necessarily a complete, independently feasible plan: the host must handle
+coupled actions and escalate the missing decisions. Exit 0 alone is not permission to execute every candidate action.
+
+Persona/live commands have command-specific verdicts. In particular, live exit 4 means a held writer lock,
+a paused individual or a retired individual; do not retry it as though it were a transient engine error.
Every recipe below was run in the cross-platform matrix (`.github/workflows/bench.yml`; transcripts in
`scripts/bench/results/`), on the OSes named under each.
@@ -55,12 +64,19 @@ node examples/node/decide.mjs router.json # decide your document
```js
import { probbit } from './decide.mjs';
const r = await probbit(['decide', '--budget-ms', '200'], problem); // problem: an object or JSON text
-if (r.kind === 'answer') act(r.output.plan, r.output.released); else escalate(r);
+if (r.kind === 'answer' && r.output.plan_status === 'released') act(r.output.released_plan);
+else escalate(r); // partial decisions need application-specific handling of coupled actions
```
The npm package (`npm/`) installs the binary and a `probbit` command with exit codes passed through. On Windows, spawn
`probbit.exe` itself (set `PROBBIT_BIN`): Node cannot spawn npm's `probbit.cmd` shim without a shell.
+## A host-enforced agent regression
+
+[examples/agent-harness](../examples/agent-harness/README.md) runs a local failing tool behind a real dispatch gate,
+records its incident, reproduces a history-dependent retry change, then checks a repaired policy and replays its strand.
+It requires no model or credentials. This is a bounded integration example, not a production permissions framework.
+
## PowerShell
Run on Windows (`windows-latest`) with PowerShell 7.6 as written.
diff --git a/docs/persona.md b/docs/persona.md
index 1ba387b..f5779df 100644
--- a/docs/persona.md
+++ b/docs/persona.md
@@ -640,6 +640,18 @@ ones it leaves unknown; the lint document gets `props` (each rule's `prove` entr
Exit codes: `fuzz` 0 nothing found, 1 a counterexample; `prove` 0 every rule held or proved, 1 some rule unknown; both 2 bad
input. JSON: `--json` (`probbit_persona_fuzz: 1`, `probbit_persona_prove: 1`); timing goes to stderr.
+**Source-restricted fixtures.** For a persona with `reward_from`, exported abstract counterexamples have no runnable
+`replay` or `explain` command unless a synthetic source is explicitly authorized with
+`--fixture-src env:synthetic` (or `human:synthetic`, where allowed). The exporter strictly replays that labelled script;
+it rejects incompatible source rules, records `fixture_provenance`, and never bypasses the production source check.
+The Python `persona_fuzz` and MCP fuzz tool accept `fixture_src` too. See the
+[host regression example](../examples/agent-harness/README.md). Synthetic observations are not production evidence.
+
+**Output formatting.** Single-document persona commands accept `--pretty` for stdout: init, turn, diff, lint, check,
+compile, describe, fuzz and prove (for fuzz/prove it implies `--json`). Written states and programs remain canonical;
+replay remains JSONL and explain remains text, so those commands reject `--pretty` with guidance. The
+`probbit_persona_turn` field is the document schema version, not its zero-based `turn` counter.
+
### 5.7 Live: a resident individual
`probbit live` keeps one individual running: JSONL events in (one object of inputs per line, from `--events FILE` or stdin),
@@ -770,7 +782,11 @@ against the replay (its `"checkpoints": N` is in the summary); `verify --from-ch
checkpoint against the event line before it (prev, the stance digest, the state digest, the state read as any state is) and
replays only what follows (`"from_checkpoint": n`); the lines before it are trusted, so run a full `verify` to check them.
`monitor` starts at the last checkpoint and draws its event at once. A strand that ends in a checkpoint line is continued from
-the state it carries. Measured on an Apple M4: a 10,000-event strand of a drives persona (4 goals, a floor, learning; 3,126,444
+the state it carries. Before continuation or a control append, the stored chain and lifecycle are checked from the header:
+a checkpoint cannot hide a retired status, an invalid transition or a broken preceding hash link. This structural scan
+costs time proportional to the existing log; it does not recompute earlier inference. Full `verify` is still necessary to
+check that inference. Hashes alone do not authenticate a writer or detect a wholly rewritten log without a trusted external
+anchor. Historical measurement before this hardening, on an Apple M4: a 10,000-event strand of a drives persona (4 goals, a floor, learning; 3,126,444
bytes, 10 checkpoint lines) opens in `monitor --once` from its last checkpoint in 0.005 s (load 3.3), where a full `verify`
takes 192 s. A replay from a checkpoint costs what the events after it cost: 500 events past the last one took 26 s to draw
(load 6-13, about 50 ms an event for this persona), so K bounds the wait; a smaller K costs a few KB per checkpoint line. A strand without checkpoint or control lines (shorter than K events, or `--checkpoint-every 0`) is the
@@ -793,6 +809,7 @@ or names the line that differs.
probbit monitor pip.strand --follow # bars in the terminal, redrawn as the strand grows
probbit monitor pip.strand --open # the same board as a page at http://127.0.0.1:PORT/, in the browser
probbit monitor --demo --open # the tutor's week, paced, for someone without a strand yet
+probbit monitor --demo drives --open # eight synthetic events with goals and drives
```
**What it shows.** The latest event. A header: persona and version, seed, the individual's digest, the event number, the hours
@@ -832,6 +849,9 @@ The page draws the terminal's board in a dark theme, the bars moving as the odds
**Demo.** `probbit monitor --demo` replays the week of `probbit live examples/persona/tutor.yaml --seed 2 --demo week` (section
5.7; the week is written to a temporary file, read back and removed), paced 1 s per hour and each night in 2 s, under a minute:
at a terminal, or with `--serve`, over and over. `--demo --once`, or a pipe, prints its last frame.
+`--demo week` explicitly selects the same tutor demo; `--demo drives` selects the embedded Scout persona's eight-event
+drive demo. Both are labelled synthetic, with a loop number and restart/completion cues. The default tutor events and
+canonical replay documents are unchanged. Demo bars are scripted engine state, not measurements of an attached agent.
Exit codes: 0 when every line replays, 1 at a line that differs (the frame names it and shows the event before it), 2 for a bad
flag, a port that cannot be had, or a file that cannot be read or is not a strand.
diff --git a/docs/probbit-ir-json.md b/docs/probbit-ir-json.md
index 67630be..eec5180 100644
--- a/docs/probbit-ir-json.md
+++ b/docs/probbit-ir-json.md
@@ -137,7 +137,10 @@ returns `Err(String)` for structural errors (lengths, out-of-range indices, `i =
weight difference beyond ~745 already decides (e^745 is past the range of double-precision probability ratios), so the limit
removes no distribution you can express; it keeps every log-space quantity (log w sums over variables, pairs and
same-group pairs; log Z) below ~1e9 x (input size), far inside the double range (1.8e308). Caps and limits are integers
- 0..2^53. Above the limit a value is rejected, not clamped: rescale your scores.
+ 0..min(2^53, usize::MAX). On wasm32 the maximum is 4,294,967,295; size/count flags must also fit that target's
+ `usize`. The sum of weights in each linear cap must fit `usize` even when individual weights are valid.
+ Seeds remain 64-bit with the JSON exact-integer bound of 2^53. Out-of-range values are rejected, never wrapped
+ or silently clamped: rescale the model or use a wider native target.
- **Arithmetic** stays in log space: Gibbs conditionals and enumeration subtract the running max before `exp`; the frontier DP
carries log weights and merges by log-sum-exp (before 0.2.0 it used linear weights scaled per layer, which underflowed past ~745
nats; with weights of +-1000 it returned NaN marginals labelled `exact` and the CLI aborted, exit 134: probbit-ir test
diff --git a/examples/agent-harness/README.md b/examples/agent-harness/README.md
new file mode 100644
index 0000000..b36390a
--- /dev/null
+++ b/examples/agent-harness/README.md
@@ -0,0 +1,72 @@
+# A recorded incident becomes an agent regression
+
+This small host runs a deliberately failing **local** probe. No model, credentials, network,
+package install, or real external action is involved. Python 3.9+ and the built `probbit`
+binary are the only requirements.
+
+From the repository root:
+
+```sh
+cargo build --release -p probbit-cli
+python3 examples/agent-harness/run.py --binary ./target/release/probbit
+PROBBIT_BIN=./target/release/probbit python3 -m unittest discover -s examples/agent-harness -v
+```
+
+Use `--out /path/to/new-directory` to retain the strands, host action receipts, exact replay
+copies, input events and proof document. Without it the demonstration uses temporary files.
+On Windows, pass the path to `target/release/probbit.exe`; set `PROBBIT_BIN` using your shell's
+environment syntax when running tests.
+
+## What runs
+
+1. The fresh `adaptive.json` policy dispatches two failing probes, then selects `ask`.
+2. Twenty **explicitly synthetic** feedback events change its bounded learned weights. The same
+ failures now allow a third attempt. The independent host limit stops further dispatch.
+3. `guarded.json` adds one hard habit: after two consecutive errors, select `ask`. The same
+ feedback and incident now dispatch only two calls; the host never executes a third.
+4. Every incident replays into a separate strand with identical bytes and no tool calls. The
+ original incident is also replayed under the repaired policy as a regression. `persona prove`
+ checks the declared stance rule for seeds 0–19 and reports `held_by_construction`.
+
+The expected tool-call counts are **2 → 3 → 2**. These are a deterministic fixture, not a
+production reliability or behavior-quality benchmark. Recorded durations vary between runs;
+replay uses the actual captured elapsed-time inputs.
+
+## The host enforces the decision
+
+`run.py:gate` dispatches only the allowlisted `local_probe` when the stance is `ok`, the
+`action` is released and equals `retry`, the habit violation count is zero, no inputs were
+ignored, and the host attempt budget remains. Partial, refused, fallback, malformed and
+unknown actions escalate without calling the tool. Engine errors abort before dispatch.
+This is a real branch around a callable—not an instruction pasted into model prose.
+
+The host measures tool outcomes and elapsed time. `host-actions.json` correlates each proposal,
+dispatch/stop, tool outcome and duration with the stance turn, state digest and strand head.
+That action receipt is separate from the engine strand; the strand verifies decisions, not
+external actions. There are no live approvals, distributed locks, crash-atomic dispatch,
+exactly-once execution, or production authentication in this deliberately bounded example.
+A plain circuit breaker can enforce this single static retry rule. The additional workflow
+shown here is history-dependent reproduction, bounded learning, deterministic replay and a
+scoped rule check.
+
+## Replaying generated counterexamples
+
+When `reward_from` requires a source, fuzz searches an abstract accepted-source history.
+It must not silently label that history as a production observation. Explicitly authorize
+a synthetic fixture source when exporting a runnable counterexample:
+
+```sh
+./target/release/probbit persona fuzz examples/agent-harness/adaptive.json --seeds 0-19 --never '{when: {error_streak: 2}, then: {action: [ask]}}' --fixture-src env:synthetic --json
+```
+
+The command exits 1 when it finds a counterexample, 0 when none was found, and 2 on an input
+error. `human:synthetic` is also accepted, but only where the persona permits human reward
+sources. No source restriction is bypassed. The exported script and final stance are produced
+by strict replay, and `fixture_provenance` records the synthetic assumption. Without the flag,
+a source-required counterexample still reports the abstract script but its `replay` and
+`explain` commands are `null`. Personas without `reward_from` keep their existing output.
+
+Do not copy synthetic labels onto production logs. In production the host must authenticate,
+classify and record original observations. A source label or hash chain alone is not proof
+of origin, and a stance proof says nothing about model obedience, subjective experience, or
+whether the author's chosen rule covers every operational hazard.
diff --git a/examples/agent-harness/adaptive.json b/examples/agent-harness/adaptive.json
new file mode 100644
index 0000000..22c7194
--- /dev/null
+++ b/examples/agent-harness/adaptive.json
@@ -0,0 +1,60 @@
+{
+ "probbit_persona": 1,
+ "identity": {
+ "name": "Retry policy",
+ "version": "1",
+ "seed": 0
+ },
+ "traits": [
+ {
+ "id": "action",
+ "levels": [
+ "retry",
+ "ask"
+ ],
+ "logw": [
+ 0.5,
+ 0
+ ],
+ "say": [
+ "retry the local probe",
+ "escalate to the operator"
+ ]
+ }
+ ],
+ "inputs": [
+ {
+ "id": "error",
+ "kind": "flag"
+ },
+ {
+ "id": "praise",
+ "kind": "flag"
+ }
+ ],
+ "history": [
+ {
+ "id": "error_streak",
+ "of": "error",
+ "kind": "streak",
+ "cap": 3,
+ "effects": {
+ "action": 0.2
+ }
+ }
+ ],
+ "learning": {
+ "from": [
+ "praise"
+ ],
+ "traits": [
+ "action"
+ ],
+ "rate": 0.3,
+ "step_cap": 0.05,
+ "total_cap": 0.4
+ },
+ "reward_from": [
+ "env"
+ ]
+}
diff --git a/examples/agent-harness/guarded.json b/examples/agent-harness/guarded.json
new file mode 100644
index 0000000..542b6ea
--- /dev/null
+++ b/examples/agent-harness/guarded.json
@@ -0,0 +1,74 @@
+{
+ "probbit_persona": 1,
+ "identity": {
+ "name": "Retry policy",
+ "version": "1",
+ "seed": 0
+ },
+ "traits": [
+ {
+ "id": "action",
+ "levels": [
+ "retry",
+ "ask"
+ ],
+ "logw": [
+ 0.5,
+ 0
+ ],
+ "say": [
+ "retry the local probe",
+ "escalate to the operator"
+ ]
+ }
+ ],
+ "inputs": [
+ {
+ "id": "error",
+ "kind": "flag"
+ },
+ {
+ "id": "praise",
+ "kind": "flag"
+ }
+ ],
+ "history": [
+ {
+ "id": "error_streak",
+ "of": "error",
+ "kind": "streak",
+ "cap": 3,
+ "effects": {
+ "action": 0.2
+ }
+ }
+ ],
+ "learning": {
+ "from": [
+ "praise"
+ ],
+ "traits": [
+ "action"
+ ],
+ "rate": 0.3,
+ "step_cap": 0.05,
+ "total_cap": 0.4
+ },
+ "reward_from": [
+ "env"
+ ],
+ "habits": [
+ {
+ "id": "escalate_after_two_errors",
+ "when": {
+ "error_streak": 2
+ },
+ "then": {
+ "action": [
+ "ask"
+ ]
+ },
+ "say": "two failures: stop and escalate"
+ }
+ ]
+}
diff --git a/examples/agent-harness/run.py b/examples/agent-harness/run.py
new file mode 100644
index 0000000..19d8119
--- /dev/null
+++ b/examples/agent-harness/run.py
@@ -0,0 +1,145 @@
+#!/usr/bin/env python3
+"""Local, dependency-free incident-to-regression example. No model or network calls."""
+import argparse
+import json
+from pathlib import Path
+import subprocess
+import sys
+import tempfile
+import time
+
+HERE = Path(__file__).resolve().parent
+sys.path.insert(0, str(HERE.parent.parent / "python"))
+import probbit
+
+RULE = {"when": {"error_streak": 2}, "then": {"action": ["ask"]}}
+
+
+class FixtureFailure(Exception):
+ pass
+
+
+class LocalProbe:
+ """Deliberately fails, without any external side effect."""
+ def __init__(self):
+ self.calls = 0
+
+ def __call__(self):
+ self.calls += 1
+ raise FixtureFailure("deliberate local fixture failure")
+
+
+def gate(stance, proposal, attempts, max_attempts=3):
+ """The host is authoritative: only a released, clean retry dispatches the allowlisted tool."""
+ if proposal != "local_probe":
+ return "escalate", "tool is not allowlisted"
+ if attempts >= max_attempts:
+ return "escalate", "independent host attempt limit"
+ if not isinstance(stance, dict) or stance.get("status") != "ok":
+ return "escalate", "stance not ok"
+ if stance.get("ignored") or stance.get("escalate"):
+ return "escalate", "ignored input or engine escalation"
+ habits = stance.get("habits")
+ if not isinstance(habits, dict) or habits.get("violations") != 0:
+ return "escalate", "habit verification unavailable or violated"
+ traits = stance.get("stance")
+ action = traits.get("action") if isinstance(traits, dict) else None
+ if not isinstance(action, dict) or action.get("released") is not True:
+ return "escalate", "action not released"
+ if action.get("level") != "retry":
+ return "escalate", "policy selected escalation"
+ return "dispatch", "released retry"
+
+
+def command(binary, *args):
+ process = subprocess.run([binary, *map(str, args)], capture_output=True, encoding="utf-8", timeout=30)
+ if process.returncode != 0:
+ raise RuntimeError("command failed: " + process.stdout + process.stderr)
+ return process.stdout
+
+
+def incident(policy, directory, binary, feedback=0):
+ """Host observations become the strand inputs; model prose is neither consulted nor executed."""
+ directory.mkdir(parents=True)
+ document = json.loads(policy.read_text(encoding="utf-8"))
+ strand = directory / "incident.strand"
+ result = probbit.live_event(document, event={}, seed=0, strand=strand, binary=binary)
+ for _ in range(feedback):
+ # A clearly labelled test assumption, not production ratings or model self-reward.
+ result = probbit.live_event(document, result["state"], {"praise": True, "src": "env:synthetic"}, strand=strand, binary=binary)
+ tool = LocalProbe()
+ actions = []
+ while True:
+ decision, reason = gate(result["stance"], "local_probe", tool.calls)
+ record = {"proposal": "local_probe", "decision": decision, "reason": reason,
+ "stance_turn": result["stance"]["turn"], "state_digest": result["state"]["digest"],
+ "strand_head": result["strand"]["head"], "attempts_before": tool.calls}
+ actions.append(record)
+ if decision != "dispatch":
+ break
+ started = time.monotonic_ns()
+ try:
+ tool()
+ except FixtureFailure as error:
+ elapsed_ns = time.monotonic_ns() - started
+ record.update(outcome="failure", duration_ns=elapsed_ns, detail=str(error))
+ result = probbit.live_event(document, result["state"],
+ {"error": True, "src": "env:local_probe", "elapsed_hours": elapsed_ns / 3.6e12}, strand=strand, binary=binary)
+ else:
+ raise AssertionError("the demonstration probe always fails")
+ (directory / "host-actions.json").write_text(json.dumps(actions, indent=2) + "\n", encoding="utf-8")
+ verified = probbit.live_verify(strand, binary=binary)
+ if not verified["ok"]:
+ raise AssertionError(verified)
+ # Replay exactly the recorded inputs into a separate strand, with no tool calls at all.
+ records = [json.loads(line) for line in strand.read_text(encoding="utf-8").splitlines()]
+ events = [record["inputs"] for record in records[1:] if "n" in record]
+ event_file = directory / "events.jsonl"
+ event_file.write_text("".join(json.dumps(event) + "\n" for event in events), encoding="utf-8")
+ copy = directory / "replayed.strand"
+ command(binary, "live", policy, "--seed", "0", "--clock", "fixed", "--events", event_file, "--strand", copy)
+ if copy.read_bytes() != strand.read_bytes():
+ raise AssertionError("incident replay changed strand bytes")
+ return {"tool_calls": tool.calls, "stop": actions[-1]["reason"], "replay_identical": True,
+ "events": verified["events"], "final_action": result["stance"]["stance"]["action"]["level"]}, events
+
+
+def demonstrate(output, binary=None):
+ binary = probbit.find_binary(binary)
+ fresh, _ = incident(HERE / "adaptive.json", output / "fresh", binary)
+ learned, events = incident(HERE / "adaptive.json", output / "learned", binary, feedback=20)
+ guarded, _ = incident(HERE / "guarded.json", output / "guarded", binary, feedback=20)
+ replay = probbit.persona_replay(HERE / "guarded.json", events, seed=0, binary=binary)
+ # Events include actual tool failures. The same history becomes a checked regression.
+ failures = [stance for stance in replay if stance["inputs"].get("error")]
+ if failures[1]["stance"]["action"]["level"] != "ask":
+ raise AssertionError("the incident regression was not repaired")
+ proof = probbit.persona_prove(HERE / "guarded.json", never=RULE, seeds="0-19", threads=1, binary=binary)
+ if proof["properties"][0]["verdict"] != "held_by_construction":
+ raise AssertionError(proof)
+ if (fresh["tool_calls"], learned["tool_calls"], guarded["tool_calls"]) != (2, 3, 2):
+ raise AssertionError((fresh, learned, guarded))
+ report = {"fresh": fresh, "after_synthetic_feedback": learned, "guarded_after_feedback": guarded,
+ "incident_regression": "passed", "stance_rule": proof["properties"][0]["verdict"],
+ "scope": "declared stance rule plus this local host gate; not model obedience, source authentication, or general agent safety"}
+ (output / "report.json").write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8")
+ (output / "proof.json").write_text(json.dumps(proof, indent=2) + "\n", encoding="utf-8")
+ return report
+
+
+def main():
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument("--binary", help="probbit executable (otherwise PROBBIT_BIN/PATH/build fallback)")
+ parser.add_argument("--out", type=Path, help="new directory for the incident, actions and replay artifacts")
+ args = parser.parse_args()
+ if args.out:
+ args.out.mkdir(parents=True, exist_ok=False)
+ report = demonstrate(args.out, args.binary)
+ else:
+ with tempfile.TemporaryDirectory(prefix="probbit-agent-example-") as directory:
+ report = demonstrate(Path(directory), args.binary)
+ print(json.dumps(report, indent=2))
+
+
+if __name__ == "__main__":
+ main()
diff --git a/examples/agent-harness/test_harness.py b/examples/agent-harness/test_harness.py
new file mode 100644
index 0000000..136239e
--- /dev/null
+++ b/examples/agent-harness/test_harness.py
@@ -0,0 +1,38 @@
+"""Run: python3 -m unittest discover -s examples/agent-harness -v."""
+import copy
+from pathlib import Path
+import tempfile
+import unittest
+from run import demonstrate, gate
+
+
+class HostEnforcement(unittest.TestCase):
+ def test_incident_replay_and_retry_repair(self):
+ with tempfile.TemporaryDirectory() as directory:
+ report = demonstrate(Path(directory))
+ self.assertEqual(report["fresh"]["tool_calls"], 2)
+ self.assertEqual(report["after_synthetic_feedback"]["tool_calls"], 3)
+ self.assertEqual(report["guarded_after_feedback"]["tool_calls"], 2)
+ self.assertEqual(report["incident_regression"], "passed")
+ self.assertTrue(all(report[key]["replay_identical"] for key in ("fresh", "after_synthetic_feedback", "guarded_after_feedback")))
+
+ def test_host_rejects_unreleased_refused_or_unknown_tool_actions(self):
+ stance = {"status": "ok", "ignored": [], "habits": {"violations": 0},
+ "stance": {"action": {"level": "retry", "released": True}}}
+ self.assertEqual(gate(stance, "local_probe", 0)[0], "dispatch")
+ cases = [None, {}, dict(stance, status="partial"), dict(stance, status="refused"),
+ dict(stance, status="fallback"), dict(stance, ignored=["typo"]),
+ dict(stance, escalate="ask"), dict(stance, habits={"violations": 1}),
+ dict(stance, habits=None), dict(stance, stance=[]), dict(stance, stance={"action": None})]
+ for key, value in (("released", False), ("level", "ask"), ("level", "arbitrary_command")):
+ changed = copy.deepcopy(stance)
+ changed["stance"]["action"][key] = value
+ cases.append(changed)
+ for result in cases:
+ self.assertEqual(gate(result, "local_probe", 0)[0], "escalate", result)
+ self.assertEqual(gate(stance, "shell", 0)[0], "escalate")
+ self.assertEqual(gate(stance, "local_probe", 3)[0], "escalate")
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/examples/node/decide.mjs b/examples/node/decide.mjs
index 600cc9d..89873d5 100644
--- a/examples/node/decide.mjs
+++ b/examples/node/decide.mjs
@@ -11,7 +11,7 @@ const KIND = { 0: 'answer', 1: 'infeasible', 2: 'bad_input', 3: 'refused' };
/**
* Run `probbit ` with `input` (an object or JSON text; omit for none) on stdin. Resolves to {kind, exitCode, output, stderr}:
- * 'answer' exit 0: verdict exact | diagnostics_passed | partial (act on output.released, escalate output.escalated)
+ * 'answer' exit 0: verdict exact | diagnostics_passed | partial (inspect released_plan and escalate escalated)
* 'infeasible' exit 1: no plan satisfies the rules (a proof, not an error)
* 'bad_input' exit 2: output.error = {code, path, message} for a bad document; a bad flag leaves only a stderr line
* 'refused' exit 3: verdict refused / declined (escalate), or output.error.code === 'numeric'
@@ -47,9 +47,9 @@ export function probbit(args, input, { bin = process.env.PROBBIT_BIN || 'probbit
function describe(r) {
const o = r.output || {};
if (r.kind === 'bad_input') return o.error ? `${o.error.code} at ${o.error.path}: ${o.error.message}` : r.stderr;
- const plan = o.plan ? Object.entries(o.plan).slice(0, 3).map(([t, w]) => `${t}->${w}`).join(', ') : '';
+ const plan = o.released_plan ? Object.entries(o.released_plan).slice(0, 3).map(([t, w]) => `${t}->${w}`).join(', ') : '';
const released = Array.isArray(o.released) ? `, ${o.released.length} of ${o.tasks} released` : '';
- return `verdict ${o.verdict}${released}${plan ? `, plan ${plan}, ...` : ''}${o.reason ? ` (${o.reason})` : ''}`;
+ return `verdict ${o.verdict}${released}${plan ? `, released assignments ${plan}, ...` : ''}${o.plan_status === 'diagnostic' ? ', candidate is diagnostic only' : ''}${o.reason ? ` (${o.reason})` : ''}`;
}
async function main() {
@@ -84,4 +84,4 @@ async function main() {
process.exit(ok ? 0 : 1);
}
-if (import.meta.url === pathToFileURL(process.argv[1]).href) main();
+if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) main();
diff --git a/install.ps1 b/install.ps1
index afc5c53..1d34815 100644
--- a/install.ps1
+++ b/install.ps1
@@ -42,6 +42,8 @@ if (-not $Version) {
if (-not $Version) { throw 'probbit install: no published release found (set -Version)' }
}
if (-not $Version.StartsWith('v')) { $Version = "v$Version" }
+if ($Version -notmatch '^v[A-Za-z0-9][A-Za-z0-9._+-]*$') { throw "probbit install: invalid release version: $Version" }
+if ($Target -notmatch '^[A-Za-z0-9_-]+$') { throw "probbit install: invalid target: $Target" }
if (-not $InstallDir) { $InstallDir = Join-Path $HOME '.local\bin' }
$name = "probbit-$Version-$Target"
@@ -64,11 +66,20 @@ try {
Expand-Archive -LiteralPath $zip -DestinationPath (Join-Path $tmp 'x') -Force
$src = Join-Path $tmp "x\$name\probbit.exe"
if (-not (Test-Path -LiteralPath $src)) { throw "probbit install: $asset has no $name\probbit.exe" }
+ # Validate before touching an existing install. A wrong-platform or damaged executable must not replace it.
+ $ran = & $src version
+ if ($LASTEXITCODE -ne 0) { throw "probbit install: downloaded binary does not run here; existing install unchanged: $ran" }
New-Item -ItemType Directory -Force -Path $InstallDir | Out-Null
$dest = Join-Path $InstallDir 'probbit.exe'
- Copy-Item -LiteralPath $src -Destination $dest -Force
- $ran = & $dest version
- if ($LASTEXITCODE -ne 0) { throw "probbit install: installed $dest, but it does not run here: $ran" }
+ $candidate = Join-Path $InstallDir ('.probbit-' + [guid]::NewGuid().ToString('N') + '.exe')
+ try {
+ Copy-Item -LiteralPath $src -Destination $candidate
+ # PowerShell coerces $null to an empty string for a string parameter; File.Replace rejects that path.
+ if (Test-Path -LiteralPath $dest) { [IO.File]::Replace($candidate, $dest, [NullString]::Value) }
+ else { [IO.File]::Move($candidate, $dest) }
+ } finally {
+ Remove-Item -LiteralPath $candidate -Force -ErrorAction SilentlyContinue
+ }
Write-Host " installed $dest ($ran)"
} finally {
Remove-Item -LiteralPath $tmp -Recurse -Force -ErrorAction SilentlyContinue
diff --git a/install.sh b/install.sh
index 8c3e8f2..df2bb36 100755
--- a/install.sh
+++ b/install.sh
@@ -6,7 +6,7 @@
#
# Environment (all optional):
# PROBBIT_VERSION release tag, e.g. v0.8.0 (default: the latest release)
-# PROBBIT_INSTALL_DIR where `probbit` goes (default: /usr/local/bin if you can write there, else ~/.local/bin)
+# PROBBIT_INSTALL_DIR where `probbit` goes (default: ~/.local/bin; never overwrites a system-wide install implicitly)
# PROBBIT_DOWNLOAD_BASE the archive is fetched from $PROBBIT_DOWNLOAD_BASE//probbit--.tar.gz
# (default: https://github.com/BitmapAsset/probbit/releases/download); needs PROBBIT_VERSION
# PROBBIT_TARGET Rust target triple to fetch instead of the detected one
@@ -54,11 +54,15 @@ detect_target() {
esac
;;
Linux)
- if (ldd --version 2>&1 || true) | grep -qi musl; then
- die "this system uses musl libc (Alpine?) and the release ships a glibc build only; build from source: cargo install --git https://github.com/$REPO probbit-cli"
- fi
case "$arch" in
- x86_64 | amd64) echo x86_64-unknown-linux-gnu ;;
+ x86_64 | amd64)
+ if (ldd --version 2>&1 || true) | grep -qi musl; then
+ echo x86_64-unknown-linux-musl
+ else
+ echo x86_64-unknown-linux-gnu
+ fi
+ ;;
+ aarch64 | arm64) echo aarch64-unknown-linux-musl ;;
*) die "no prebuilt probbit for Linux on $arch yet; build from source: cargo install --git https://github.com/$REPO probbit-cli" ;;
esac
;;
@@ -91,6 +95,8 @@ main() {
[ -n "$version" ] || die "no published release found (set PROBBIT_VERSION)"
fi
case "$version" in v*) ;; *) version="v$version" ;; esac
+ case "$version" in *[!A-Za-z0-9._+-]* | v) die "invalid release version: $version" ;; esac
+ case "$target" in *[!A-Za-z0-9_-]* | '') die "invalid target: $target" ;; esac
name="probbit-$version-$target"
asset="$name.tar.gz"
@@ -109,11 +115,12 @@ main() {
tar -xzf "$tmp/$asset" -C "$tmp" || die "could not unpack $asset"
[ -f "$tmp/$name/probbit" ] || die "$asset has no $name/probbit"
+ # Check the candidate before replacing an existing working install (wrong architecture, corrupt executable, etc.).
+ chmod 755 "$tmp/$name/probbit" || die "could not make the downloaded binary executable"
+ ran=$("$tmp/$name/probbit" version 2>&1) || die "downloaded binary does not run here; existing install unchanged: $ran"
if [ -n "${PROBBIT_INSTALL_DIR:-}" ]; then
dir="$PROBBIT_INSTALL_DIR"
- elif [ -d /usr/local/bin ] && [ -w /usr/local/bin ]; then
- dir=/usr/local/bin
else
dir="${HOME:?HOME is not set; set PROBBIT_INSTALL_DIR}/.local/bin"
fi
@@ -121,7 +128,6 @@ main() {
[ -w "$dir" ] || die "cannot write to $dir (set PROBBIT_INSTALL_DIR to a directory you can write; this script never uses sudo)"
cp "$tmp/$name/probbit" "$dir/.probbit.$$" && chmod 755 "$dir/.probbit.$$" && mv -f "$dir/.probbit.$$" "$dir/probbit" \
|| die "could not install into $dir"
- ran=$("$dir/probbit" version 2>&1) || die "installed $dir/probbit, but it does not run here: $ran"
say " installed $dir/probbit ($ran)"
cmd=probbit
@@ -130,7 +136,8 @@ main() {
*)
cmd="$dir/probbit"
say ""
- say "$dir is not on your PATH. Add it (e.g. in ~/.profile):"
+ case "${SHELL:-}" in */zsh) profile='~/.zshrc' ;; */bash) profile='~/.bashrc (or ~/.bash_profile for login shells)' ;; *) profile='your shell profile' ;; esac
+ say "$dir is not on your PATH. Add it in $profile:"
say " export PATH=\"$dir:\$PATH\""
;;
esac
diff --git a/npm/README.md b/npm/README.md
index 28d62d6..af6c320 100644
--- a/npm/README.md
+++ b/npm/README.md
@@ -11,7 +11,7 @@ probbit stats --pretty
```
The install step fetches the release archive for your platform from GitHub Releases (macOS arm64 and x86_64, Linux
-x86_64 with glibc, Windows x86_64), checks it against the `.sha256` file published beside it and keeps the binary inside
+x86_64 with glibc or musl, Linux ARM64 static musl, Windows x86_64), checks it against the `.sha256` file published beside it and keeps the binary inside
this package. No dependencies; Node >= 18. If the install step could not run (`--ignore-scripts`, offline), the first
`probbit` call fetches the binary. A checksum mismatch fails the install and installs nothing.
diff --git a/npm/bin/probbit.js b/npm/bin/probbit.js
index 5e1d66e..b7da35d 100755
--- a/npm/bin/probbit.js
+++ b/npm/bin/probbit.js
@@ -16,15 +16,16 @@ async function main() {
}
const child = spawn(bin, process.argv.slice(2), { stdio: 'inherit' });
const signals = ['SIGINT', 'SIGTERM', 'SIGHUP'].filter((s) => process.platform !== 'win32' || s !== 'SIGHUP');
- const forward = (s) => { try { child.kill(s); } catch (e) { /* already gone */ } };
- for (const s of signals) process.on(s, forward);
+ // Node's signal events carry no argument. Bind each signal explicitly instead of accidentally forwarding SIGTERM.
+ const handlers = new Map(signals.map(s => [s, () => { try { child.kill(s); } catch (e) { /* already gone */ } }]));
+ for (const [s, handler] of handlers) process.on(s, handler);
child.on('error', (e) => {
process.stderr.write(`probbit: cannot run ${bin}: ${e.message}\n`);
process.exit(127);
});
child.on('exit', (code, signal) => {
if (signal) {
- for (const s of signals) process.removeListener(s, forward);
+ for (const [s, handler] of handlers) process.removeListener(s, handler);
process.exitCode = 128 + (os.constants.signals[signal] || 0); // if the signal is ignored here (SIGPIPE)
process.kill(process.pid, signal); // die the same way, so the shell sees 128 + n
return;
diff --git a/npm/install.js b/npm/install.js
index 2d38ae4..42c5730 100644
--- a/npm/install.js
+++ b/npm/install.js
@@ -22,6 +22,7 @@ const TARGETS = {
'darwin-arm64': 'aarch64-apple-darwin',
'darwin-x64': 'x86_64-apple-darwin',
'linux-x64': 'x86_64-unknown-linux-gnu',
+ 'linux-arm64': 'aarch64-unknown-linux-musl',
'win32-x64': 'x86_64-pc-windows-msvc',
'win32-arm64': 'x86_64-pc-windows-msvc', // x64 emulation on Windows on Arm
};
@@ -34,7 +35,7 @@ function binaryPath() {
class ChecksumError extends Error {}
async function get(url) {
- const res = await fetch(url, { redirect: 'follow' });
+ const res = await fetch(url, { redirect: 'follow', signal: AbortSignal.timeout(30000) });
if (!res.ok) throw new Error(`GET ${url}: HTTP ${res.status}`);
return Buffer.from(await res.arrayBuffer());
}
@@ -49,7 +50,8 @@ function place(src, dest) {
fs.renameSync(tmp, dest);
} catch (e) {
fs.rmSync(tmp, { force: true });
- if (!fs.existsSync(dest)) throw e; // another process put it there first
+ // Only forgive an actual identical concurrent install, not an old binary left behind by a permission/rename error.
+ if (!fs.existsSync(dest) || !fs.readFileSync(src).equals(fs.readFileSync(dest))) throw e;
}
}
@@ -62,13 +64,18 @@ async function install(log = (m) => process.stderr.write(`${m}\n`)) {
log(`probbit: using ${src} (PROBBIT_BINARY)`);
return dest;
}
- const target = env.PROBBIT_TARGET || TARGETS[`${process.platform}-${process.arch}`];
+ const glibc = process.platform === 'linux' && process.report && process.report.getReport().header.glibcVersionRuntime;
+ const detected = process.platform === 'linux' && process.arch === 'x64' && !glibc
+ ? 'x86_64-unknown-linux-musl' : TARGETS[`${process.platform}-${process.arch}`];
+ const target = env.PROBBIT_TARGET || detected;
if (!target) {
throw new Error(`no prebuilt probbit for ${process.platform}-${process.arch}; build one (cargo build --release -p probbit-cli) `
+ 'and reinstall with PROBBIT_BINARY=/path/to/probbit');
}
let tag = env.PROBBIT_VERSION || `v${pkg.version}`;
if (!tag.startsWith('v')) tag = `v${tag}`;
+ if (!/^v[A-Za-z0-9][A-Za-z0-9._+-]*$/.test(tag)) throw new Error(`invalid release version: ${tag}`);
+ if (!/^[A-Za-z0-9_-]+$/.test(target)) throw new Error(`invalid target: ${target}`);
const base = (env.PROBBIT_DOWNLOAD_BASE || `https://github.com/${REPO}/releases/download`).replace(/\/+$/, '');
const zip = target.includes('windows');
const name = `probbit-${tag}-${target}`;
diff --git a/npm/test-wrapper.cjs b/npm/test-wrapper.cjs
new file mode 100644
index 0000000..52aaacb
--- /dev/null
+++ b/npm/test-wrapper.cjs
@@ -0,0 +1,92 @@
+'use strict';
+// Dependency-free unit checks for platform routing and native-process contracts.
+const { test } = require('node:test');
+const assert = require('node:assert/strict');
+const fs = require('node:fs');
+const path = require('node:path');
+const vm = require('node:vm');
+const { EventEmitter } = require('node:events');
+
+function installer(platform, arch, glibc, env = {}) {
+ const urls = [];
+ const module = { exports: {} };
+ const context = {
+ require, module, __dirname, Buffer, AbortSignal,
+ process: { platform, arch, env, report: { getReport: () => ({ header: { glibcVersionRuntime: glibc } }) } },
+ fetch: async url => { urls.push(url); throw new Error('fixture: network disabled'); },
+ };
+ vm.runInNewContext(fs.readFileSync(path.join(__dirname, 'install.js'), 'utf8'), context);
+ return { ...module.exports, urls };
+}
+
+for (const [platform, arch, glibc, target] of [
+ ['darwin', 'arm64', null, 'aarch64-apple-darwin'],
+ ['darwin', 'x64', null, 'x86_64-apple-darwin'],
+ ['linux', 'x64', '2.35', 'x86_64-unknown-linux-gnu'],
+ ['linux', 'x64', null, 'x86_64-unknown-linux-musl'],
+ ['linux', 'arm64', '2.35', 'aarch64-unknown-linux-musl'],
+ ['win32', 'x64', null, 'x86_64-pc-windows-msvc'],
+ ['win32', 'arm64', null, 'x86_64-pc-windows-msvc'],
+]) {
+ test(`archive selection: ${platform}/${arch}/${glibc || 'no glibc'}`, async () => {
+ const fixture = installer(platform, arch, glibc);
+ await assert.rejects(fixture.install(() => {}), /network disabled/);
+ assert.ok(fixture.urls[0].includes(`-${target}.`), fixture.urls[0]);
+ assert.equal(fixture.urls.length, 2);
+ });
+}
+
+test('unsupported architecture fails without fetching', async () => {
+ const fixture = installer('linux', 'riscv64', null);
+ await assert.rejects(fixture.install(() => {}), /no prebuilt/);
+ assert.equal(fixture.urls.length, 0);
+});
+
+for (const env of [{ PROBBIT_VERSION: '../bad' }, { PROBBIT_TARGET: '../../bad' }]) {
+ test(`invalid download selector: ${Object.keys(env)[0]}`, async () => {
+ const fixture = installer('darwin', 'arm64', null, env);
+ await assert.rejects(fixture.install(() => {}), /invalid/);
+ assert.equal(fixture.urls.length, 0);
+ });
+}
+
+function wrapper(platform = 'linux') {
+ const child = new EventEmitter();
+ const signals = [];
+ child.kill = signal => signals.push(signal);
+ const process = new EventEmitter();
+ Object.assign(process, { env: { PROBBIT_BINARY: '/fixture/probbit' }, argv: ['node', 'wrapper', 'monitor', '--follow'],
+ platform, pid: 42, stderr: { write() {} }, exit(code) { this.exitCode = code; }, kill(pid, signal) { this.killed = [pid, signal]; } });
+ let invocation;
+ const requireFixture = id => {
+ if (id === 'child_process') return { spawn: (binary, args, opts) => { invocation = { binary, args: [...args], stdio: opts.stdio }; return child; } };
+ if (id === '../install.js') return { binaryPath: () => '/unused', install: () => { throw new Error('unexpected download'); } };
+ return require(id);
+ };
+ vm.runInNewContext(fs.readFileSync(path.join(__dirname, 'bin/probbit.js'), 'utf8'), { require: requireFixture, process });
+ return { child, process, signals, invocation };
+}
+
+test('wrapper preserves arguments, streams, and every documented exit code', () => {
+ for (const code of [0, 1, 2, 3, 4]) {
+ const fixture = wrapper();
+ assert.deepEqual(fixture.invocation, { binary: '/fixture/probbit', args: ['monitor', '--follow'], stdio: 'inherit' });
+ fixture.child.emit('exit', code, null);
+ assert.equal(fixture.process.exitCode, code);
+ }
+});
+
+test('wrapper forwards the exact signal even though Node signal events carry no arguments', () => {
+ const fixture = wrapper();
+ for (const signal of ['SIGINT', 'SIGTERM', 'SIGHUP']) fixture.process.emit(signal);
+ assert.deepEqual(fixture.signals, ['SIGINT', 'SIGTERM', 'SIGHUP']);
+ fixture.child.emit('exit', null, 'SIGINT');
+ assert.deepEqual(fixture.process.killed, [42, 'SIGINT']);
+ assert.equal(fixture.process.listenerCount('SIGINT'), 0);
+});
+
+test('wrapper reports native spawn failures as 127', () => {
+ const fixture = wrapper();
+ fixture.child.emit('error', new Error('ENOENT'));
+ assert.equal(fixture.process.exitCode, 127);
+});
diff --git a/playground/check-browser.mjs b/playground/check-browser.mjs
new file mode 100644
index 0000000..fbd53b9
--- /dev/null
+++ b/playground/check-browser.mjs
@@ -0,0 +1,139 @@
+// Real Chromium UI checks, with no npm dependencies. Node >=22 and Chrome/Chromium are test-only requirements.
+// Build first: sh playground/build.sh. Then: node playground/check-browser.mjs [screenshot-directory]
+import assert from 'node:assert/strict';
+import { spawn, spawnSync } from 'node:child_process';
+import { createServer } from 'node:http';
+import { existsSync, mkdtempSync, mkdirSync, readFileSync, writeFileSync, rmSync } from 'node:fs';
+import { tmpdir } from 'node:os';
+import { dirname, extname, join, resolve, sep } from 'node:path';
+import { fileURLToPath } from 'node:url';
+import { once } from 'node:events';
+
+const root = resolve(dirname(fileURLToPath(import.meta.url)), '..');
+const profile = mkdtempSync(join(tmpdir(), 'probbit-browser-'));
+const output = resolve(process.argv[2] || mkdtempSync(join(tmpdir(), 'probbit-browser-shots-')));
+mkdirSync(output, { recursive: true });
+const executable = process.env.CHROME_BIN || [
+ '/Applications/Google Chrome.app/Contents/MacOS/Google Chrome',
+ '/usr/bin/google-chrome', '/usr/bin/chromium', '/usr/bin/chromium-browser',
+].find(existsSync) || spawnSync('which', ['google-chrome'], { encoding: 'utf8' }).stdout.trim();
+assert.ok(executable, 'Set CHROME_BIN to an installed Chrome or Chromium executable');
+assert.equal(typeof WebSocket, 'function', 'This test requires Node >=22 (built-in WebSocket)');
+
+const server = createServer((req, res) => {
+ const pathname = decodeURIComponent(new URL(req.url, 'http://localhost').pathname);
+ const file = resolve(root, '.' + pathname);
+ if (!file.startsWith(root + sep)) { res.writeHead(403); res.end(); return; }
+ try {
+ const body = readFileSync(file);
+ const type = { '.html': 'text/html', '.js': 'text/javascript', '.wasm': 'application/wasm', '.json': 'application/json' }[extname(file)] || 'application/octet-stream';
+ res.writeHead(200, { 'Content-Type': type }); res.end(body);
+ } catch { res.writeHead(404); res.end('Not found'); }
+});
+server.listen(0, '127.0.0.1');
+await once(server, 'listening');
+const base = `http://127.0.0.1:${server.address().port}`;
+const chrome = spawn(executable, ['--headless=new', '--no-first-run', '--no-default-browser-check',
+ '--disable-background-networking', '--remote-debugging-port=0', `--user-data-dir=${profile}`, 'about:blank'],
+{ stdio: ['ignore', 'ignore', 'pipe'] });
+let socket;
+const pending = new Map();
+const errors = [];
+let nextId = 0;
+try {
+ const url = await new Promise((ok, fail) => {
+ const timer = setTimeout(() => fail(new Error('Chrome debugger did not start within 20 seconds')), 20000);
+ let stderr = '';
+ chrome.on('error', e => { clearTimeout(timer); fail(e); });
+ chrome.on('exit', code => { clearTimeout(timer); fail(new Error(`Chrome exited: ${code}\n${stderr}`)); });
+ chrome.stderr.on('data', b => { stderr += b; const match = stderr.match(/DevTools listening on (ws:\/\/\S+)/); if (match) { clearTimeout(timer); ok(match[1]); } });
+ });
+ socket = new WebSocket(url);
+ await new Promise((ok, fail) => { socket.addEventListener('open', ok, { once: true }); socket.addEventListener('error', fail, { once: true }); });
+ socket.addEventListener('message', event => {
+ const msg = JSON.parse(event.data);
+ if (msg.method === 'Runtime.exceptionThrown') errors.push(msg.params.exceptionDetails);
+ if (msg.id && pending.has(msg.id)) {
+ const { ok, fail, timer } = pending.get(msg.id); pending.delete(msg.id); clearTimeout(timer);
+ if (msg.error) fail(new Error(JSON.stringify(msg.error))); else ok(msg.result);
+ }
+ });
+ function call(method, params = {}, sessionId) {
+ const id = ++nextId;
+ return new Promise((ok, fail) => {
+ const timer = setTimeout(() => { pending.delete(id); fail(new Error(`CDP timeout: ${method}`)); }, 30000);
+ pending.set(id, { ok, fail, timer });
+ socket.send(JSON.stringify({ id, method, params, ...(sessionId ? { sessionId } : {}) }));
+ });
+ }
+ const { targetId } = await call('Target.createTarget', { url: 'about:blank' });
+ const { sessionId } = await call('Target.attachToTarget', { targetId, flatten: true });
+ const page = (method, params) => call(method, params, sessionId);
+ await page('Runtime.enable');
+ await page('Page.enable');
+ async function evaluate(expression) {
+ const r = await page('Runtime.evaluate', { expression, returnByValue: true, awaitPromise: true });
+ if (r.exceptionDetails) throw new Error(JSON.stringify(r.exceptionDetails));
+ return r.result.value;
+ }
+ async function until(expression) {
+ const deadline = Date.now() + 25000;
+ while (Date.now() < deadline) {
+ if (await evaluate(expression)) return;
+ await new Promise(r => setTimeout(r, 100));
+ }
+ throw new Error(`UI timeout: ${expression}`);
+ }
+ async function screenshot(name) {
+ const shot = await page('Page.captureScreenshot', { format: 'png', captureBeyondViewport: false });
+ writeFileSync(join(output, name + '.png'), Buffer.from(shot.data, 'base64'));
+ }
+ async function viewport(width) {
+ await page('Emulation.setDeviceMetricsOverride', { width, height: 1000, deviceScaleFactor: 1, mobile: false });
+ }
+ await viewport(1440);
+ await page('Page.navigate', { url: base + '/playground/index.html' });
+ await until('document.title === "probbit playground: ready"');
+ let result = await evaluate('JSON.parse(document.querySelector("#raw").textContent)');
+ assert.equal(result.verdict, 'exact'); assert.equal(result.violations, 0);
+ assert.equal(result.answers.team.probbit.value, 'technical');
+ assert.match(await evaluate('document.querySelector("#meet-status").textContent'), /turns/);
+ await screenshot('playground-desktop');
+ // Drive the public controls, including an error followed by recovery.
+ await evaluate('document.querySelector("#input").value = "{bad"; document.querySelector("#go").click()');
+ await until('!document.querySelector("#go").disabled && document.querySelector("#verdict").textContent.startsWith("ERROR")');
+ await evaluate('document.querySelector("#ex-demo").click(); document.querySelector("#go").click()');
+ await until('!document.querySelector("#go").disabled && JSON.parse(document.querySelector("#raw").textContent).tasks === 300');
+ result = await evaluate('JSON.parse(document.querySelector("#raw").textContent)');
+ assert.equal(result.verdict, 'diagnostics_passed'); assert.equal(result.violations, 0);
+ await viewport(390);
+ await screenshot('playground-mobile');
+ const playgroundWidth = await evaluate('({page: document.documentElement.scrollWidth, viewport: innerWidth})');
+ assert.ok(playgroundWidth.page <= playgroundWidth.viewport + 1, `Playground overflow: ${JSON.stringify(playgroundWidth)}`);
+ console.log(JSON.stringify({ check: 'playground browser controls, errors, recovery, personas, 390px layout', passed: true }));
+
+ await viewport(1440);
+ await page('Page.navigate', { url: base + '/playground/puzzle.html' });
+ await until('document.querySelector("#lineA")?.textContent.includes("Stance:")');
+ await evaluate('document.querySelector("#reveal").click(); document.querySelector("#replay").click()');
+ await until('document.querySelector("#replayOut").textContent.includes("same")');
+ const replay = await evaluate('document.querySelector("#replayOut").textContent');
+ assert.match(replay, /Both branches repeat byte for byte and equal the native CLI's pinned values/);
+ assert.equal(await evaluate('document.querySelectorAll("#replayOut .bad").length'), 0);
+ await screenshot('puzzle-desktop');
+ await viewport(390);
+ await screenshot('puzzle-mobile');
+ const puzzleWidth = await evaluate('({page: document.documentElement.scrollWidth, viewport: innerWidth})');
+ assert.ok(puzzleWidth.page <= puzzleWidth.viewport + 1, `Puzzle overflow: ${JSON.stringify(puzzleWidth)}`);
+ assert.equal(errors.length, 0, JSON.stringify(errors));
+ console.log(JSON.stringify({ check: 'puzzle browser reveal and replay, 390px layout, no uncaught JS errors', passed: true }));
+ console.log(JSON.stringify({ screenshots: output, browser: (await call('Browser.getVersion')).product }));
+} finally {
+ for (const { timer } of pending.values()) clearTimeout(timer);
+ if (socket) socket.close();
+ chrome.kill('SIGTERM');
+ await new Promise(resolve => { if (chrome.exitCode !== null) return resolve(); const timer = setTimeout(() => { chrome.kill('SIGKILL'); resolve(); }, 3000); chrome.once('exit', () => { clearTimeout(timer); resolve(); }); });
+ server.closeAllConnections();
+ await new Promise(resolve => server.close(resolve));
+ rmSync(profile, { recursive: true, force: true });
+}
diff --git a/playground/index.html b/playground/index.html
index 0a556ac..5d5ff3d 100644
--- a/playground/index.html
+++ b/playground/index.html
@@ -20,6 +20,9 @@
#verdict { margin: 10px 0; font-size: 18px; font-weight: 700; letter-spacing: 0.04em; }
.exact, .diagnostics_passed { color: var(--green); } .partial { color: var(--yellow); } .refused, .declined, .infeasible, .error { color: var(--red); }
table { border-collapse: collapse; font: 13px/1.4 ui-monospace, SFMono-Regular, Menlo, monospace; margin-top: 6px; }
+ .table-scroll { max-width: 100%; overflow-x: auto; }
+ .table-scroll:focus-visible { outline: 2px solid var(--cyan); outline-offset: 3px; }
+ #runs, #verdict, .note { overflow-wrap: anywhere; }
th, td { text-align: left; padding: 4px 12px 4px 0; border-bottom: 1px solid var(--line); vertical-align: top; } th { color: var(--dim); font-weight: 500; }
td.moved { color: var(--magenta); } .note { color: var(--dim); font-size: 13px; margin-top: 6px; }
details { margin-top: 16px; } summary { cursor: pointer; color: var(--dim); }
@@ -39,7 +42,7 @@
probbitplayground: WebAssembly, one thread, no server
-
+
full answer (JSON)
Meet three individuals from one persona
@@ -52,7 +55,7 @@
Meet three individuals from one persona
seeds
-
+
Each cell: the stance line (habits first, then traits away from this individual's resting level), then three traits' levels with
their exact odds. Habits hold in every cell (violations 0 by construction); the three differ where their genes do.
diff --git a/playground/puzzle.html b/playground/puzzle.html
index 2bc7f37..6fd322c 100644
--- a/playground/puzzle.html
+++ b/playground/puzzle.html
@@ -17,6 +17,7 @@
h1 small { display: block; font-size: 14px; font-weight: 400; color: var(--dim); letter-spacing: 0.02em; }
h2 { margin: 30px 0 8px; font-size: 18px; }
p { max-width: 860px; }
+ .panel, .muted, .line { overflow-wrap: anywhere; min-width: 0; }
.sub { color: var(--dim); margin: 10px 0 16px; }
.bar { display: flex; flex-wrap: wrap; gap: 8px; align-items: center; margin: 10px 0; }
button, select, input { background: var(--panel); color: var(--text); border: 1px solid var(--line); border-radius: 6px; padding: 7px 12px; font: inherit; }
@@ -84,6 +85,8 @@
.caption strong { color: var(--text); } .caption span { color: var(--dim); }
@media (max-width: 560px) {
body { font-size: 14px; }
+ table { table-layout: fixed; }
+ th, td { overflow-wrap: anywhere; }
.cards { grid-template-columns: repeat(2, minmax(0, 1fr)); }
.card { min-height: 60px; }
.lv { font-size: 12.5px; }
diff --git a/probbit-cli/src/fuzz.rs b/probbit-cli/src/fuzz.rs
index d03a6ab..8c974c3 100644
--- a/probbit-cli/src/fuzz.rs
+++ b/probbit-cli/src/fuzz.rs
@@ -107,6 +107,44 @@ pub fn fuzz(p: &Persona, props: &[Prop], s: &Search, eng: SyncEngine) -> (Vec>], source: &str, eng: persona::Engine) -> Result<(), crate::json::InErr> {
+ if !matches!(source, "env:synthetic" | "human:synthetic") {
+ return Err(persona::perr("fixture_src", "env:synthetic | human:synthetic; an explicit test-source assumption, never production provenance"));
+ }
+ if !persona::has_reward_sources(p) { return Err(persona::perr("fixture_src", "requires a persona with reward_from")); }
+ for per in res {
+ for (seed, f) in seeds.iter().zip(per) {
+ let Some(f) = f else { continue };
+ let mut script = f.script.clone();
+ for event in &mut script {
+ if event.iter().any(|(k, _)| k == "src") { return Err(persona::perr("fixture_src", "will not replace an existing source")); }
+ event.push(("src".into(), Json::Str(source.into())));
+ persona::check_provenance(p, &persona::event_json(event))?;
+ }
+ let events: Vec = script.iter().map(|e| persona::event_json(e)).collect();
+ let (stances, _) = persona::replay(p, Some(*seed), &events, false, eng, false)?;
+ f.doc = stances.last().expect("a counterexample has an event").clone();
+ f.script = script;
+ }
+ }
+ Ok(())
+}
+fn replayable(p: &Persona, f: &Found) -> bool {
+ f.script.iter().all(|e| persona::check_provenance(p, &persona::event_json(e)).is_ok())
+}
+fn fixture_info(p: &Persona, f: &Found) -> Vec<(String, Json)> {
+ if !persona::has_reward_sources(p) { return vec![]; }
+ let source = f.script.first().and_then(|e| e.iter().find(|(k, _)| k == "src")).map(|(_, v)| v.clone()).unwrap_or(Json::Null);
+ vec![("fixture_provenance".into(), Json::Obj(vec![
+ ("source".into(), source.clone()), ("synthetic".into(), Json::Bool(true)),
+ ("replayable".into(), Json::Bool(replayable(p, f))),
+ ("assumption".into(), Json::Str(if source.is_null() {
+ "search assumes accepted sources; use --fixture-src env:synthetic or human:synthetic to authorize synthetic replay; production evidence must retain its original source"
+ } else { "caller-authorized synthetic fixture source, not authenticated production evidence" }.into()))]))]
+}
/// `f(0) .. f(n-1)` on up to `threads` threads (each takes the next index; 8 MiB stacks, as the engine's threads), in index order
fn par(n: usize, threads: usize, f: impl Fn(usize) -> T + Sync) -> Vec {
let next = AtomicUsize::new(0); let slots: Vec>> = (0..n).map(|_| Mutex::new(None)).collect();
@@ -187,9 +225,10 @@ pub fn ranges(seeds: &[u64]) -> String {
out.join(",")
}
fn script_json(sc: &[Event]) -> Json { Json::Arr(sc.iter().map(|e| persona::event_json(e)).collect()) }
-fn commands(path: &str, seed: u64, f: &Found) -> (String, String) {
+fn commands(p: &Persona, path: &str, seed: u64, f: &Found) -> Option<(String, String)> {
+ if !replayable(p, f) { return None; }
let sc = word(&persona::canon(&script_json(&f.script)));
- (format!("probbit persona replay {} --seed {seed} --script {sc}", word(path)), format!("probbit persona explain {} --seed {seed} --script {sc} --turn {}", word(path), f.script.len() - 1))
+ Some((format!("probbit persona replay {} --seed {seed} --script {sc}", word(path)), format!("probbit persona explain {} --seed {seed} --script {sc} --turn {}", word(path), f.script.len() - 1)))
}
fn lengths(per: &[Option]) -> Vec<(usize, usize)> {
let mut by: Vec<(usize, usize)> = vec![];
@@ -208,10 +247,11 @@ pub fn doc(p: &Persona, path: &str, props: &[Prop], s: &Search, res: &[Vec = props.iter().zip(res).map(|(pr, per)| {
let failing: Vec = s.seeds.iter().zip(per).filter(|(_, f)| f.is_some()).map(|(sd, _)| n(*sd as f64)).collect();
- let sh = shortest(s, per).map_or(Json::Null, |(sd, f)| { let (rp, ex) = commands(path, sd, f);
- Json::Obj(vec![("seed".into(), n(sd as f64)), ("script".into(), script_json(&f.script)), ("turn".into(), n((f.script.len() - 1) as f64)), ("broken".into(), broken_json(&f.broken)),
- ("stance".into(), f.doc.clone()), ("replay".into(), st(&rp)), ("explain".into(), st(&ex))]) });
- let cex: Vec = s.seeds.iter().zip(per).filter_map(|(sd, f)| f.as_ref().map(|f| Json::Obj(vec![("seed".into(), n(*sd as f64)), ("script".into(), script_json(&f.script))]))).collect();
+ let sh = shortest(s, per).map_or(Json::Null, |(sd, f)| { let cmds = commands(p, path, sd, f);
+ let (rp, ex) = cmds.map_or((Json::Null, Json::Null), |(r, e)| (st(&r), st(&e)));
+ let mut fields = vec![("seed".into(), n(sd as f64)), ("script".into(), script_json(&f.script)), ("turn".into(), n((f.script.len() - 1) as f64)), ("broken".into(), broken_json(&f.broken)),
+ ("stance".into(), f.doc.clone()), ("replay".into(), rp), ("explain".into(), ex)]; fields.extend(fixture_info(p, f)); Json::Obj(fields) });
+ let cex: Vec = s.seeds.iter().zip(per).filter_map(|(sd, f)| f.as_ref().map(|f| { let mut fields = vec![("seed".into(), n(*sd as f64)), ("script".into(), script_json(&f.script))]; fields.extend(fixture_info(p, f)); Json::Obj(fields) })).collect();
Json::Obj(vec![("id".into(), st(&pr.id)), ("rule".into(), pr.rule.clone()), ("verdict".into(), st(if failing.is_empty() { "none_found" } else { "counterexample" })),
("individuals".into(), n(s.seeds.len() as f64)), ("failing".into(), n(failing.len() as f64)), ("failing_seeds".into(), Json::Arr(failing)),
("by_length".into(), Json::Obj(lengths(per).into_iter().map(|(l, c)| (l.to_string(), n(c as f64))).collect())), ("shortest".into(), sh), ("counterexamples".into(), Json::Arr(cex))]) }).collect();
@@ -246,7 +286,9 @@ pub fn human(p: &Persona, path: &str, props: &[Prop], s: &Search, res: &[Vec = s.seeds.iter().zip(per).filter(|(_, f)| f.is_some()).map(|(x, _)| *x).collect();
o.push(format!(" failing seeds: {}", ranges(&seeds))); }
}
@@ -274,7 +316,7 @@ pub fn seeds(v: &str) -> Option> {
pub fn tool(args: &[(String, Json)], eng: SyncEngine) -> Result {
use persona::perr;
let get = |k: &str| args.iter().find(|(x, _)| x == k).map(|(_, v)| v).filter(|v| !v.is_null());
- const KNOWN: [&str; 12] = ["persona", "persona_path", "never", "props", "seeds", "fuzz_seed", "scripts", "depth", "beam", "grid", "hours", "threads"];
+ const KNOWN: [&str; 13] = ["persona", "persona_path", "never", "props", "seeds", "fuzz_seed", "scripts", "depth", "beam", "grid", "hours", "threads", "fixture_src"];
let mut extra: Vec<&str> = args.iter().map(|(k, _)| k.as_str()).filter(|k| !KNOWN.contains(k)).collect(); extra.sort_unstable();
if let Some(k) = extra.first() { return Err(perr(&format!("arguments.{k}"), "unknown argument")); }
let (p, path) = match (get("persona"), get("persona_path")) {
@@ -301,7 +343,11 @@ pub fn tool(args: &[(String, Json)], eng: SyncEngine) -> Result Json { persona::run_program(prog, f, 1, 100, 0) }
+ #[test]
+ fn synthetic_fixture_export_replays_strictly_without_weakening_sources() {
+ let p = persona::build(&json::parse(r#"{"probbit_persona":1,"identity":{"name":"Fixture","version":"1"},
+ "traits":[{"id":"action","levels":["retry","ask"],"logw":[1,0]}],
+ "inputs":[{"id":"praise","kind":"flag"},{"id":"criticism","kind":"flag"}],
+ "learning":{"from":["praise","criticism"],"traits":["action"],"rate":3,"step_cap":2,"total_cap":2},
+ "reward_from":["env"]}"#).unwrap()).unwrap();
+ let prs = persona::props(&p, &json::parse(r#"{"then":{"action":["retry"]}}"#).unwrap(), "never", "never").unwrap();
+ let s = Search { seeds: vec![0], fuzz_seed: 0, scripts: 0, depth: 3, beam: 4, grid: vec![0.0, 1.0], hours: vec![1.0], threads: 1 };
+ let (mut res, turns) = fuzz(&p, &prs, &s, &run);
+ let before = doc(&p, "fixture.json", &prs, &s, &res, turns);
+ let shortest = before.get("properties").unwrap().as_arr().unwrap()[0].get("shortest").unwrap();
+ assert_eq!(shortest.get("replay"), Some(&Json::Null));
+ let events = shortest.get("script").unwrap().as_arr().unwrap();
+ assert!(persona::replay(&p, Some(0), events, false, &run, false).is_err());
+ for src in ["env:production", "human:synthetic", "self", "clock"] {
+ assert!(authorize_fixtures(&p, &s.seeds, &mut res, src, &run).is_err(), "{src}");
+ }
+ authorize_fixtures(&p, &s.seeds, &mut res, "env:synthetic", &run).unwrap();
+ let f = res[0][0].as_ref().unwrap();
+ let events = script_json(&f.script);
+ let (replayed, _) = persona::replay(&p, Some(0), events.as_arr().unwrap(), false, &run, false).unwrap();
+ assert_eq!(replayed.last(), Some(&f.doc));
+ assert!(commands(&p, "fixture.json", 0, f).is_some());
+ assert!(authorize_fixtures(&p, &s.seeds, &mut res, "env:synthetic", &run).is_err(), "existing labels are never replaced");
+ let state = persona::init(&p, Some(0), true, &run);
+ for ev in [r#"{"praise":true}"#, r#"{"praise":true,"src":"self"}"#] {
+ assert!(persona::turn(&p, &state, &json::parse(ev).unwrap(), false, &run, false).is_err());
+ }
+ }
+
/// A habit of a generated persona: id, `when`, `then` (JSON text) and priority
struct H { id: String, when: String, then: String, priority: usize }
/// A random persona: 2-3 traits (2-3 levels), a mood (inertia 0.3-0.8) coupled to each, flags f0 and f1, a level input lv (neg,
diff --git a/probbit-cli/src/json.rs b/probbit-cli/src/json.rs
index 021d0b9..9022ffc 100644
--- a/probbit-cli/src/json.rs
+++ b/probbit-cli/src/json.rs
@@ -63,7 +63,7 @@ pub const MAX_WEIGHT: f64 = 1e9;
/// Above it: a `limit` error before anything is allocated. The largest documented program is 200 x 65,535 = 13.1 million.
pub const MAX_DENSE: usize = 20_000_000;
pub fn weight(j: &Json, path: &str) -> Result {
- let x = number(j, path)?; if x.abs() > MAX_WEIGHT { return Err(limit(path, format!("{x:e} is beyond the weight limit |x| <= 1e9 (natural-log odds; rescale)"))); } Ok(x)
+ let x = number(j, path)?; if !x.is_finite() || x.abs() > MAX_WEIGHT { return Err(limit(path, format!("{x:e} is beyond the finite weight limit |x| <= 1e9 (natural-log odds; rescale)"))); } Ok(x)
}
/// Paths of the non-finite numbers in `j` (a decision may carry none; main.rs `finish`).
pub fn non_finite(j: &Json, path: &str, out: &mut Vec) {
@@ -71,9 +71,14 @@ pub fn non_finite(j: &Json, path: &str, out: &mut Vec) {
Json::Obj(v) => for (k, x) in v { non_finite(x, &at(path, k), out) }, _ => {} }
}
/// A whole number in 0..=2^53 (exact in a double): caps and limits.
-pub fn count(j: &Json, path: &str) -> Result {
+pub fn count_u64(j: &Json, path: &str) -> Result {
let x = number(j, path)?; if x < 0.0 || x.fract() != 0.0 { return Err(value(path, "must be a non-negative integer")); }
- if x > 9_007_199_254_740_992.0 { return Err(limit(path, "must be at most 2^53")); } Ok(x as usize)
+ if x > 9_007_199_254_740_992.0 { return Err(limit(path, "must be at most 2^53")); } Ok(x as u64)
+}
+/// A count that also fits this platform (wasm32 must not silently saturate it).
+pub fn count(j: &Json, path: &str) -> Result {
+ let x = count_u64(j, path)?;
+ usize::try_from(x).map_err(|_| limit(path, "integer exceeds this platform's size limit"))
}
/// An array of distinct strings (`values`, `allowed`, `forbid`, cap `vars`).
pub fn names<'a>(j: &'a Json, path: &str) -> Result, InErr> {
@@ -103,8 +108,24 @@ fn node(b: &[u8], i: &mut usize, d: usize) -> Result {
b't' if b[*i..].starts_with(b"true") => { *i += 4; Ok(Json::Bool(true)) }
b'f' if b[*i..].starts_with(b"false") => { *i += 5; Ok(Json::Bool(false)) }
b'n' if b[*i..].starts_with(b"null") => { *i += 4; Ok(Json::Null) }
- b'-' | b'0'..=b'9' => { let s = *i; *i += 1;
- while *i < b.len() && matches!(b[*i], b'0'..=b'9' | b'.' | b'e' | b'E' | b'+' | b'-') { *i += 1; }
+ b'-' | b'0'..=b'9' => { let s = *i;
+ // JSON's number grammar is narrower than Rust's f64 parser (which
+ // also accepts 01, 1., and -.1). Require each digit group explicitly.
+ if b[*i] == b'-' { *i += 1; }
+ match b.get(*i) {
+ Some(b'0') => *i += 1,
+ Some(b'1'..=b'9') => { *i += 1; while b.get(*i).is_some_and(u8::is_ascii_digit) { *i += 1; } }
+ _ => return Err(format!("bad number at byte {s}: expected an integer part")),
+ }
+ if b.get(*i) == Some(&b'.') { *i += 1; let start = *i;
+ while b.get(*i).is_some_and(u8::is_ascii_digit) { *i += 1; }
+ if *i == start { return Err(format!("bad number at byte {s}: expected fractional digits")); }
+ }
+ if matches!(b.get(*i), Some(b'e' | b'E')) { *i += 1;
+ if matches!(b.get(*i), Some(b'+' | b'-')) { *i += 1; } let start = *i;
+ while b.get(*i).is_some_and(u8::is_ascii_digit) { *i += 1; }
+ if *i == start { return Err(format!("bad number at byte {s}: expected exponent digits")); }
+ }
let x = std::str::from_utf8(&b[s..*i]).unwrap().parse::().map_err(|e| format!("bad number at byte {s}: {e}"))?;
if !x.is_finite() { return Err(format!("limit: number {} at byte {s} is not a finite double", String::from_utf8_lossy(&b[s..*i]))); } Ok(Json::Num(x)) }
c => Err(format!("unexpected character '{}' at byte {i}", c as char)),
@@ -124,7 +145,9 @@ fn string(b: &[u8], i: &mut usize) -> Result {
if (0xDC00..0xE000).contains(&lo) { cp = 0x10000 + ((cp - 0xD800) << 10) + (lo - 0xDC00); } else { *i = s; } }
out.push(char::from_u32(cp).ok_or_else(|| format!("lone surrogate \\u{cp:04x} at byte {}", *i - 6))?); }
_ => return Err(format!("bad escape at byte {i}")) } }
- _ => { let s = *i; while *i < b.len() && b[*i] != b'"' && b[*i] != b'\\' { *i += 1; } out.push_str(std::str::from_utf8(&b[s..*i]).map_err(|e| e.to_string())?); } } }
+ _ => { let s = *i; while *i < b.len() && b[*i] != b'"' && b[*i] != b'\\' {
+ if b[*i] < 0x20 { return Err(format!("unescaped control character at byte {i}")); } *i += 1;
+ } out.push_str(std::str::from_utf8(&b[s..*i]).map_err(|e| e.to_string())?); } } }
}
/// The 4 hex digits of a `\u` escape at `*i` (advanced past them).
diff --git a/probbit-cli/src/live.rs b/probbit-cli/src/live.rs
index be04ea0..4506fed 100644
--- a/probbit-cli/src/live.rs
+++ b/probbit-cli/src/live.rs
@@ -131,7 +131,7 @@ pub fn utc_now() -> String {
/// rules. `open` checks the header and rebuilds the individual; each `step` replays one event line on the fixed clock and
/// checks it (`prev`, `n`, the stance and state digests, the bytes). After an error the replay is spent: the chain past the
/// line that differs cannot be checked.
-pub struct Replay { pub live: Live, pub header: Json, pub doc: Json, pub line: usize, pub from: Option, last: Option }
+pub struct Replay { pub live: Live, pub header: Json, pub doc: Json, pub line: usize, pub from: Option, last: Option, checkpoint_ready: bool }
impl Replay {
/// The header line (without its line ending) -> the replay, ready for event lines; Err((1, what differs))
pub fn open(head: &str) -> Result {
@@ -143,7 +143,7 @@ impl Replay {
let st0 = State::read(&p, h.get("state").unwrap_or(&Json::Null)).map_err(|e| (1, format!("the initial state: {}: {}", e.path, e.msg)))?;
let (live, header) = Live::start(p, &doc, st0, Clock::Fixed, h.get("engine").and_then(Json::as_str).unwrap_or(""));
if header != head { return Err((1, "the header is not as written".into())); }
- Ok(Replay { live, header: h, doc, line: 1, from: None, last: None })
+ Ok(Replay { live, header: h, doc, line: 1, from: None, last: None, checkpoint_ready: false })
}
/// The replay from a checkpoint line (`cp`, without its line ending, at 1-based line `at`; `before` = the event line before it)
/// instead of the header: the header is read as by `open`; the checkpoint must follow `before` (prev), at its event count, with
@@ -154,6 +154,12 @@ impl Replay {
let j = json::parse(cp).map_err(|e| (at, format!("not JSON: {}", e.msg)))?;
let b = json::parse(before).map_err(|e| (at - 1, format!("not JSON: {}", e.msg)))?;
if persona::canon(&j) != cp { return Err((at, "not canonical JSON".into())); }
+ if !j.as_obj().is_some_and(|kv| kv.len() == 4 && kv.iter().all(|(k, _)| ["checkpoint", "prev", "stance", "state"].contains(&k.as_str()))) {
+ return Err((at, "checkpoint fields must be checkpoint, prev, stance and state".into()));
+ }
+ if !b.as_obj().is_some_and(|kv| kv.len() == 5 && kv.iter().all(|(k, _)| ["n", "inputs", "prev", "stance", "state"].contains(&k.as_str()))) {
+ return Err((at - 1, "a checkpoint must immediately follow an event line".into()));
+ }
if j.get("prev").and_then(Json::as_str) != Some(persona::digest_of(before).as_str()) { return Err((at, "prev is not the sha256 of the line before".into())); }
let n = j.get("checkpoint").and_then(Json::as_f64).filter(|x| *x >= 1.0 && x.fract() == 0.0).ok_or((at, "not a checkpoint line".to_string()))? as u64;
if b.get("n").and_then(Json::as_f64) != Some(n as f64) { return Err((at, format!("the line before is not event {n}"))); }
@@ -161,14 +167,16 @@ impl Replay {
if b.get("stance").and_then(Json::as_str) != Some(persona::sha(&stance).as_str()) { return Err((at, "the checkpoint's stance is not the one event {n} logs".replace("{n}", &n.to_string()))); }
let st = State::read(&r.live.p, j.get("state").unwrap_or(&Json::Null)).map_err(|e| (at, format!("the checkpoint's state: {}: {}", e.path, e.msg)))?;
if b.get("state").and_then(Json::as_str) != Some(st.digest.as_str()) { return Err((at, format!("the checkpoint's state is not the one event {n} logs"))); }
- r.live.st = st; r.live.n = n; r.live.prev = persona::digest_of(cp); r.live.checkpoints = 1; r.line = at; r.from = Some(n); r.last = Some(stance);
+ r.live.st = st; r.live.n = n; r.live.prev = persona::digest_of(cp); r.live.checkpoints = 1; r.line = at; r.from = Some(n); r.last = Some(stance); r.checkpoint_ready = false;
Ok(r)
}
/// The stance document of the last event replayed (or of the checkpoint the replay started from)
pub fn last_stance(&self) -> Option<&Json> { self.last.as_ref() }
/// One line (without its line ending) -> the stance document an event line replays to (None for a control or checkpoint
/// line); Err((its 1-based line number, what differs)). A control line must be a valid move of the status and the line its
- /// fields give; a checkpoint line must carry the event count and the state the replay reached.
+ /// fields give; a checkpoint line must immediately follow an event while active and carry
+ /// the event count and the state the replay reached. The last stance remains available to
+ /// the monitor after controls/checkpoints, but that does not authorize another checkpoint.
pub fn step(&mut self, l: &str, eng: Engine) -> Result