diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 7c79968..9c978bd 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -3,8 +3,18 @@ on: push: branches: [main] pull_request: +permissions: + contents: read jobs: + msrv: + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - uses: actions/checkout@v7 + - uses: dtolnay/rust-toolchain@1.78.0 + - run: cargo build --release --locked --workspace test: + timeout-minutes: 30 strategy: fail-fast: false matrix: @@ -13,6 +23,9 @@ jobs: steps: - uses: actions/checkout@v7 - uses: dtolnay/rust-toolchain@stable + - uses: actions/setup-node@v6 + with: + node-version: '24' - name: build run: cargo build --release --workspace - name: test @@ -26,13 +39,61 @@ jobs: run: cargo run --release -p probbit-cli -- stats --pretty - name: agent_router example run: cargo run --release --example agent_router + - name: Python and MCP integration contracts + shell: bash + run: | + B="$PWD/target/release/probbit" + if [ "$RUNNER_OS" = Windows ]; then B="$(cygpath -w "$B.exe")"; fi + export PROBBIT_BIN="$B" + python -m unittest discover -s python -p 'test_*.py' + node examples/node/decide.mjs + - name: npm platform and process contracts + run: node --test npm/test-wrapper.cjs + - name: clean-prefix install and npm package contracts + shell: bash + run: | + set -euo pipefail + TARGET=$(rustc -vV | sed -n 's/^host: //p') + TAG=v$(node -p "require('./npm/package.json').version") + BIN=target/release/probbit + [ "$RUNNER_OS" = Windows ] && BIN="$BIN.exe" + if [ "$RUNNER_OS" != Windows ]; then + python3 scripts/tests/test_install_contract.py + sh scripts/bench/test_install_sh.sh "$BIN" "$TARGET" "$TAG" + else + sh scripts/bench/package_like_release.sh "$BIN" "$TARGET" "$TAG" "target/install-fixture/good/$TAG" + cp -R target/install-fixture/good target/install-fixture/bad + printf '%064d %s\n' 0 "probbit-$TAG-$TARGET.zip" > "target/install-fixture/bad/$TAG/probbit-$TAG-$TARGET.zip.sha256" + for PS in pwsh powershell; do + "$PS" -NoProfile -ExecutionPolicy Bypass -File scripts/bench/test_install_ps1.ps1 -Srv target/install-fixture -Tag "$TAG" + done + fi + sh scripts/bench/test_npm.sh "$BIN" "$TARGET" "$TAG" + - name: npm 12 clean-prefix contracts + if: runner.os == 'Linux' + shell: bash + run: | + npm install -g npm@12 + sh scripts/bench/test_npm.sh target/release/probbit x86_64-unknown-linux-gnu + npm-node18: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-node@v6 + with: + node-version: '18' + - run: node --test npm/test-wrapper.cjs wasm: + timeout-minutes: 15 # probbit-wasm, the browser build (wasm32-unknown-unknown, no threads): built as playground/build.sh builds it, then loaded by # Node with the playground's own loader and run on the 300-task demo (--sweeps 3200) and the evaluate example; check.mjs # exits 1 unless both answer with 0 violations. The no-thread = 4-thread equality runs natively in `test` (probbit-wasm tests). runs-on: ubuntu-latest steps: - uses: actions/checkout@v7 + - uses: actions/setup-node@v6 + with: + node-version: '24' - uses: dtolnay/rust-toolchain@stable with: targets: wasm32-unknown-unknown @@ -40,6 +101,22 @@ jobs: run: sh playground/build.sh - name: run under Node run: node --version && node playground/check.mjs playground/probbit.wasm + - name: WASM integer boundaries and native parity + run: | + cargo build --release -p probbit-cli + node probbit-wasm/tests/boundaries.mjs playground/probbit.wasm target/release/probbit + - name: real browser interaction and narrow layout + run: node playground/check-browser.mjs target/browser-check + - name: real monitor playback and live append in the browser + env: + PROBBIT_BIN: ${{ github.workspace }}/target/release/probbit + PROBBIT_BROWSER_OUT: ${{ github.workspace }}/target/browser-check/monitor + run: node probbit-cli/tests/monitor_browser.mjs + - uses: actions/upload-artifact@v7 + if: always() + with: + name: browser-check + path: target/browser-check/ release-targets: # The release archives' targets that `test` does not build (release.yml runs only on tags): built here, so a tag is # never the first build of a target. Static musl (x86_64 runs here, aarch64 is cross-linked), macOS x86_64 under @@ -50,6 +127,7 @@ jobs: include: - { os: ubuntu-latest, target: x86_64-unknown-linux-musl, run: true } - { os: ubuntu-latest, target: aarch64-unknown-linux-musl, linker: aarch64-linux-gnu-gcc } + - { os: ubuntu-24.04-arm, target: aarch64-unknown-linux-musl, linker: aarch64-linux-gnu-gcc, run: true } - { os: macos-latest, target: x86_64-apple-darwin, run: true, rosetta: true } - { os: windows-latest, target: x86_64-pc-windows-msvc, run: true, rustflags: "-C target-feature=+crt-static" } runs-on: ${{ matrix.os }} diff --git a/.gitignore b/.gitignore index 8b6b2f7..992daf3 100644 --- a/.gitignore +++ b/.gitignore @@ -10,3 +10,28 @@ __pycache__/ playground/probbit.wasm playground/probbit-wasm.js playground/puzzle-personas.js + +# Local credentials, agent context and generated release artifacts are not product sources. +.env +.env.* +*.credentials* +*credentials.json +cookies.txt +login_response* +.secrets/ +.agents/ +.ouroboros/ +AGENTS.md +SOUL.md +USER.md +MEMORY.md +IDENTITY.md +memory/ +identity/ +diary/ +state/ +audits/ +data/ +node_modules/ +dist/ +*.tgz diff --git a/CHANGELOG.md b/CHANGELOG.md index 242da1e..216e2e5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,6 +3,39 @@ All notable changes to this project are documented here. The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/). +## Unreleased + +### Fixed +- Reject invalid JSON number grammar and raw control characters, platform-sized integer overflow, overflowing linear-cap + totals, and invalid pair-table dimensions before allocation. WASM no longer wraps large precedence gaps, counts or weights. +- Validate supplied anneal/polish starting assignments even when the work budget is zero. +- Validate stored strand hashes and lifecycle transitions before continuation or control appends; a checkpoint cannot hide + retirement. Full inference verification and external anchoring remain separate responsibilities. +- Python rejects an explicitly missing binary instead of falling back, and rejects invalid false/array input values rather + than silently treating them as empty objects. MCP and Python preserve command-specific error/lifecycle contracts. +- Persona `--pretty` consistently formats single-document stdout without changing canonical state files. JSONL replay and + text explain reject the flag with guidance. Source-restricted fuzz exports require an explicitly labelled synthetic source + to produce runnable replay commands; production source checks are not bypassed. +- npm forwards each termination signal unchanged, bounds download waits and selects both x86_64 and ARM64 musl assets. + Shell installation defaults to the user's `~/.local/bin`. Installers validate candidate binaries before replacing a + working installation on the shell/PowerShell paths; PowerShell stages replacement on the destination filesystem. + Release/target selectors reject paths. +- Monitor labels scripted demos, loop restarts, lifecycle and checkpoint status; preserves original `why` text, explains + display terms and supports narrow screens. Browser playground and puzzle tables no longer force horizontal page overflow. +- Keep the lockfile readable by the declared Rust 1.78 minimum; CI builds that compiler with `--locked`. + +### Added +- Decision documents with a candidate `plan` gain `plan_status` and `released_plan`. The old full candidate is preserved for + compatibility; a diagnostic/refused plan is not permission to act. Partial projections may not be independently feasible. +- `probbit monitor --demo drives` shows eight replayable synthetic goal events. The default tutor demo is unchanged. +- `examples/agent-harness`: a local dispatch gate with a recorded incident, history-dependent retry regression, strict + replay and a scoped repaired-rule proof. No model, credentials or external tool action required. +- Independent enumeration/router oracles, actual WASM boundary tests, real Chromium interaction/layout checks, Node process + contracts and cross-platform clean-prefix install tests (including Windows PowerShell 5.1/7 and npm 12). + +These changes are a source candidate, not an update to the published 0.8.0 artifacts. Probability gate thresholds and frozen +benchmark corpora are unchanged. Passing finite tests is not exhaustive validation of every possible program or application. + ## 0.8.0 - 2026-10-09 ### Added diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index c7374cb..360de3f 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -7,7 +7,7 @@ rules that keep its claims honest. ``` cargo build --release --workspace # no network needed: there are no external crates -cargo test --release --workspace # 127 tests: 3 core, 48 CLI, 11 stress (+1 ignored), 18 acceptance, 47 probbit-ir +cargo test --release --workspace # unit, integration, regression and exact-oracle tests cargo run --release -p probbit-cli -- demo --tasks 12 | cargo run --release -p probbit-cli -- decide --pretty ``` @@ -19,6 +19,11 @@ loaded machine; re-run it alone before reading anything into a failure. The default build is portable (no CPU pin). `RUSTFLAGS="-C target-cpu=native"` gives the last bit of speed on your own machine; that binary may not run elsewhere. +Release-contract checks also run on every pull request: the Python and MCP suites, npm platform/signal tests, +and clean-prefix shell, PowerShell 5.1/7 and npm installs using locally packaged candidate binaries. +`node --test npm/test-wrapper.cjs` needs no npm dependencies. Set `PROBBIT_BIN` to the release executable and run +`python3 -m unittest discover -s python -p 'test_*.py'` for the integration contracts. + ## Ground rules - **No external crates on the shipped path.** `cargo build` must keep working offline. diff --git a/Cargo.lock b/Cargo.lock index 6a5f8ce..ded6842 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1,6 +1,6 @@ # This file is automatically @generated by Cargo. # It is not intended for manual editing. -version = 4 +version = 3 [[package]] name = "probbit-cli" diff --git a/PORTABILITY.md b/PORTABILITY.md index 376d172..3b31faf 100644 --- a/PORTABILITY.md +++ b/PORTABILITY.md @@ -122,8 +122,9 @@ No release exists yet, so every installer was tested against a local server that | npm wrapper (`npm/`) | `ubuntu-latest`, `windows-latest`, local M4 | `npm pack`, `npm install -g` (scratch prefix), postinstall fetch and SHA-256 check, `which probbit`, exit codes 0 / 1 / 2 / 3 passed through, `npm uninstall -g`; an `--ignore-scripts` install fetches on first run; `PROBBIT_BINARY`; a wrong `.sha256` fails the install | | `docs/agents.md` recipes | shell, Python (`python/test_probbit.py`), Node (`examples/node/decide.mjs`): Linux, macOS arm64 and x86_64, Windows (Git Bash); PowerShell 7.6: Windows | each recipe as written, plus one call per exit code | -Install locations: `install.sh` writes `/usr/local/bin` when it can (every runner image above: their user can write it) -and `~/.local/bin` otherwise (the local M4, where `/usr/local/bin` belongs to root), and never uses sudo; `install.ps1` +Historical install locations in the measurements above: `install.sh` wrote `/usr/local/bin` when writable, +and `~/.local/bin` otherwise. The current source defaults to `~/.local/bin` on every shell host; use +`PROBBIT_INSTALL_DIR` for an explicit system-wide destination. It never uses sudo. `install.ps1` writes `$HOME\.local\bin`, adds it to the session's PATH, and to the user PATH only with `-AddToPath`. Names: `probbit` is free on npm and `probbit`, `probbit-core`, `probbit-ir`, `probbit-decide` and `probbit-cli` are free on crates.io (checked 2026-10-01 22:31 PDT); nothing was published. @@ -139,7 +140,8 @@ free on npm and `probbit`, `probbit-core`, `probbit-ir`, `probbit-decide` and `p 2. **The Linux release binary needs glibc 2.34.** `release.yml` builds `x86_64-unknown-linux-gnu` on `ubuntu-latest`. The static musl build above runs on any x86_64 Linux and was within 4% (run 2) and 7% (run 1) of the glibc build on the same VM, with 2.5-3.2 MiB less peak memory (BENCHMARK-MATRIX.md). Shipping it, plus `aarch64-unknown-linux-musl`, - would also give `install.sh` something for Alpine and arm64 Linux, which it refuses today. + now provides the released binaries for Alpine and arm64 Linux. The current shell/npm installers select these + targets automatically; the earlier refusal described in these historical measurements is superseded. 3. **The Windows binary needs `VCRUNTIME140.dll`.** Linking the CRT statically (`-C target-feature=+crt-static` for the msvc release build) would remove that; not built or measured here. 4. **Two wall-clock test bounds failed on shared runners** in run 1: `run_deadline_ms_bounds_the_whole_call` on diff --git a/README.md b/README.md index 5f92ff7..9d52e8e 100644 --- a/README.md +++ b/README.md @@ -118,6 +118,11 @@ were verified from clean machines on macOS (Apple silicon) and Linux (x86_64). npm may print an `allow-scripts` warning: the package's only install script downloads the prebuilt binary for your platform and checks its SHA-256. If npm blocks the script, the binary is fetched the first time you run `probbit` instead. +The shell installer defaults to `~/.local/bin`, without replacing a system-wide installation. Set +`PROBBIT_INSTALL_DIR` to choose another directory. Both the shell and npm installers select static musl binaries +for Linux ARM64 and for x86_64 musl systems such as Alpine. The shell and PowerShell installers check that a +downloaded candidate starts successfully before replacing an existing binary; failed checksums leave it unchanged. + From source, with Rust 1.78 or later: `cargo install --git https://github.com/BitmapAsset/probbit probbit-cli`, or clone and `cargo build --release -p probbit-cli` (the binary lands in `target/release/probbit`). Nothing is downloaded after the clone: there are no external crates. @@ -128,7 +133,7 @@ At a terminal, `install.sh` ends with probbit's own hero screen. Platform notes ``` git clone https://github.com/BitmapAsset/probbit && cd probbit -cargo test --release --workspace # 127 tests (3 core, 48 CLI, 11 stress, 18 acceptance, 47 probbit-ir; 1 ignored), ~25 s once built +cargo test --release --workspace # unit, integration, regression and exact-oracle tests cargo run --release -p probbit-cli -- demo --tasks 12 | cargo run --release -p probbit-cli -- decide --pretty cargo run --release --example agent_router # the full narrated demo, ~3 s ``` @@ -437,11 +442,11 @@ to be measured per model (no such measurement has been made here). ### Test a character

a corridor of glass event cards; one card cracks cyan beside a rabbit mark; a replay trace runs under the floor

-

The fuzz finds the flaw: the shortest event script that pushes an individual out of character, shrunk and replayable.

+

The fuzz finds the flaw: an event script that pushes an individual out of character, shrunk and replayable.

A character property is a rule in habit syntax the stance must never break. `fuzz` searches event scripts for each -individual's shortest counterexample; `prove` says `held by construction`, `proved for every event sequence` or `unknown`. The +individual's short counterexample (not a guaranteed global minimum); `prove` says `held by construction`, `proved for every event sequence` or `unknown`. The tutor as it shipped in 0.5.0 is kept as a fixture, so this runs from a clone with no keys and no model: ```sh @@ -529,6 +534,7 @@ anywhere ([docs/persona.md](docs/persona.md) §5.8). No strand yet? The tutor's ```sh probbit monitor --demo --open +probbit monitor --demo drives --open # source checkout: synthetic goals/drive demo, not yet in the 0.8.0 release ``` ![probbit monitor: the tutor's week in the browser, the bars moving with every event](docs/probbit-monitor.gif) diff --git a/USE-CASES.md b/USE-CASES.md index be4e75b..a4d7736 100644 --- a/USE-CASES.md +++ b/USE-CASES.md @@ -2,9 +2,12 @@ **One processor, many kinds of decision.** If you can write a decision as choices, scores and rules that must never break, it runs on the same p-bit engine: tasks to agents, steps to tools, jobs to time slots, items under a budget, cells of a -puzzle, the temperament of an agent. Every shape below gets the same three things back: a plan that keeps every rule by -construction, odds for every part (exact whenever the structure lets probbit count them), and a verdict that refuses when its +puzzle, the temperament of an agent. A feasible answer supplies a rule-abiding candidate plan, +odds (exact whenever the structure lets probbit count them), and a verdict that refuses when its diagnostics fail. And every shape gets the same honesty: where a solver, an annealer or plain backtracking wins, the table says so. +An infeasible or resource-limited problem need not produce a plan. A retained candidate on refusal is diagnostic only; +check `plan_status`, `released_plan` and the escalation set before host actions. This is a finite-domain probabilistic +decision processor, not a replacement for arbitrary CPU computation or a guarantee that any application will fit its limits. Every "gives" cell below points to a measured table in `BENCHMARKS.md` (§ numbers) or a separate run; every "do not promise" cell points to where probbit lost or was not tested. All measurements are on one Apple M4 Mac mini with synthetic or generated @@ -30,7 +33,7 @@ Do not use it when: | problem shape | how it maps to the probbit instruction set (`probbit-ir`) | what probbit gives (measured) | do not promise | who might use it *(inferred)* | |---|---|---|---|---| -| **Assign items to resources under rules** (tasks to agents/models/people, tickets to agents, jobs to machines) with quotas and a same-group preference | one categorical variable per item over its allowed resources; unary log-weights = your scores; at-most-`cap` per resource; a Potts pair per same-group pair (`probbit decide` lowers to this) | **0 rule violations by construction** (300-task demo: per-task argmax breaks 144 rules incl. 60 PII tasks sent to cloud models; probbit 0); **per-item odds you can check**: 1,731 released tasks on queues with an exact answer, 0 wrong (§2.1, current gate, re-run on 0.2.0; 1,916 on the 0.1.0 gate); **refusal/escalation** by budget (§2.3, demo 2 below); **exact odds in 0.17-28 ms** up to ~24 tasks (§2.1, §2.3); what-if clamps, including clamps that fill a worker (acceptance test `forced_caps_do_not_freeze_the_gate`: 3 queues whose clamps fill a worker + 1 forced queue each pass the gate within 0.05 TV of exact, green on the current build; a 0.1.0 record of 10/10 passed, 0 false, was not re-run) | the single best plan (ILP wins, §2.2); full coverage on saturated or strong-affinity queues (51% of tasks released on the oracle queues on the current gate, 58% on 0.1.0, §2.1); speed (the sampled path spends its ~0.3 s budget by design, §2.3) | AI platform teams routing LLM tasks; support operations; internal-tools developers | +| **Assign items to resources under rules** (tasks to agents/models/people, tickets to agents, jobs to machines) with quotas and a same-group preference | one categorical variable per item over its allowed resources; unary log-weights = your scores; at-most-`cap` per resource; a Potts pair per same-group pair (`probbit decide` lowers to this) | **0 rule violations by construction** (300-task `agent_router` example: per-task argmax breaks 144 rules incl. 60 PII tasks sent to cloud models; probbit 0); **per-item odds you can check**: 1,731 released tasks on queues with an exact answer, 0 wrong (§2.1, current gate, re-run on 0.2.0; 1,916 on the 0.1.0 gate); **refusal/escalation** by budget (§2.3, demo 2 below); **exact odds in 0.17-28 ms** up to ~24 tasks (§2.1, §2.3); what-if clamps, including clamps that fill a worker (acceptance test `forced_caps_do_not_freeze_the_gate`: 3 queues whose clamps fill a worker + 1 forced queue each pass the gate within 0.05 TV of exact, green on the current build; a 0.1.0 record of 10/10 passed, 0 false, was not re-run) | the single best plan (ILP wins, §2.2); full coverage on saturated or strong-affinity queues (51% of tasks released on the oracle queues on the current gate, 58% on 0.1.0, §2.1); speed (the sampled path spends its ~0.3 s budget by design, §2.3) | AI platform teams routing LLM tasks; support operations; internal-tools developers | | **Binary pairwise models** (Ising, max-cut, binary Markov random fields) | one two-valued variable per spin; `table` pairs = couplings; unary = fields | **per-spin odds whose gate verdict was checked against brute force**: 144 runs, 61 passed the gate, 0 false whole answers, 929 released spins, 0 wrong (§3, re-run on 0.2.0; the 0.1.0 gate: 72 passed, 1,346 released, 0 wrong); exact odds and log Z by the exact tiers on small models; at equal time, cuts within 0.06-0.7% of simulated annealing in anneal-only mode and 0.17-1.75% short when sampling (the mode that gives odds, §3) | better cuts than simulated annealing (it won by 4-7 edges, §3); log Z from the sampler (not implemented); correctness beyond 16 spins (the brute-force oracle stops there) | researchers and teachers of p-bit / Ising computing; people prototyping before renting annealer time | | **Constraint satisfaction with odds** (graph colouring, one-hot puzzles, slot assignment) | a categorical variable per node over colours/digits; negative Potts or at-most-1 caps for "differ"; clamps for givens | **exact per-variable odds and solution counts** from the exact tier (sudoku odds equal a backtracking count on 20 puzzles; colouring 0.00-5.70 ms median, §4); the sampler released 525 vertices, 0 outside tolerance (§4; 468 in an earlier build) | speed against backtracking (15-102x slower per puzzle, §4); sampler coverage on frozen or few-solution instances (sudoku: refused 20/20; 3-colouring: 9 of 14 refused on the default set in the current build, 59 of 82 over 6 sets, §4) | CSP and puzzle tooling, teaching, frequency/register-assignment prototypes *(unmeasured on real instances)* | | **Scheduling with capacity per time slot** (unit-time jobs, release/deadline windows, precedence) | a variable per job over slots; allowed = window; one cap per slot; precedence i -> j as pair caps "not (i at a and j at b)" for a >= b (O(T^2) caps per edge) | **exact odds on small instances** and a gate that released 428 jobs on 39 oracle instances, 0 outside tolerance; at 200-500 jobs, anneal from a greedy plan beat restarted greedy + repair at equal 100 ms on 4/5 seeds each (§4.3) | optimal schedules at scale (an ILP solver, HiGHS, proves the 200-job optimum in 0.68-0.95 s; probbit's plan is 3.3-4.3 nats short, §4.3); checked odds at scale (at 200 jobs the sampler needs 0.7 s just to start, refuses at 1 s, releases 153-171/200 at 5 s on 3 instances in an earlier build (re-run on a later build, N = 3: 156-173 on two, the third refused) (earlier build, start inside the budget; 2 runs each) with no oracle to check them); multi-period durations (unit jobs only) | planners prototyping small shift/slot problems *(inferred)* | @@ -61,12 +64,15 @@ Do not use it when: - **CLI + JSON** on stdin/stdout, exit codes 0 plan / 1 infeasible / 2 bad input / 3 refused. Every `bench/` script drives it. Process overhead is ~2-5 ms (§2.3: 9.0 ms wall vs 7.1 ms inside probbit at 12 tasks, 303.1 vs 299.0 at 300; 0.2.0 re-run 10.1 vs 8.0 and 314.3 vs 309.4). -- **Python**: `python/probbit.py` (0.2.0), a stdlib-only subprocess wrapper: `run` / `exact` / `sample` / `decide`, typed errors, - `deadline_ms`; 9 tests run inside `cargo test`; three examples in `python/examples/`. No native bindings (one process per call). +- **Python**: `python/probbit.py`, a stdlib-only subprocess wrapper for decisions, evaluation, personas and live events; + typed errors and deadlines. CI runs unittest discovery over `python/`. No native bindings (one process per call). - **Schema**: `docs/probbit-ir.schema.json` (JSON Schema of the probbit-ir wire format; a test keeps it equal to the parser); `probbit --help` lists every flag with its default. - **Rust**: the `probbit-decide` crate (router front-end) and `probbit-ir` (general programs, JSON v1 in `docs/probbit-ir-json.md`). -- **MCP server**: not built (README roadmap). +- **MCP server**: `probbit mcp` serves the ten documented tools over stdio; see [docs/agents.md](docs/agents.md). +- **Host-enforced incident regression**: [examples/agent-harness](examples/agent-harness/README.md) records a local tool + failure, reproduces history-dependent retries, gates dispatch and checks a repaired rule. A bounded example, not a + production authorization system. - **Running beside other work**: `--threads`, `--cpu-limit`, `--priority low`, `--mem-limit-mb`, `--progress`, `probbit stats` (§5). Reproducible runs: fixed `--sweeps` plus `--polish-sweeps`. The defaults are wall-clock (sampling and polish), so the odds, the verdict, the released set and the plan can all vary run to run and machine to machine. diff --git a/docs/agents.md b/docs/agents.md index 4a7b47c..9d31278 100644 --- a/docs/agents.md +++ b/docs/agents.md @@ -11,7 +11,16 @@ an agent harness with a shell tool. No server, no bindings, no network. The docu | 0 | an answer: verdict `exact`, `diagnostics_passed` or `partial` | act on `released`; escalate `escalated` (empty unless `partial`) | | 1 | `infeasible`: no plan satisfies the rules (a proof) | relax a rule or a cap; it is an answer, not a crash | | 2 | bad input: one `{"error": {"code", "path", "message"}}` object on stdout; a bad flag prints one line on stderr instead | fix the document or the flag | -| 3 | `refused` / `declined` (the gate or a cap said no), or `{"error": {"code": "numeric"}}` | escalate the whole decision; the best-effort plan is still in the output | +| 3 | `refused` / `declined` (the gate or a cap said no), or `{"error": {"code": "numeric"}}` | escalate the whole decision; any retained candidate plan is diagnostic only | + +Decision documents that carry a `plan` also carry `plan_status` (`released`, `partial`, or `diagnostic`) +and `released_plan`, which projects the candidate onto the released IDs. On refusal this projection is empty; +some early failures have no plan at all. The legacy `plan` remains for inspection, not unconditional execution. +A partial projection is **not** necessarily a complete, independently feasible plan: the host must handle +coupled actions and escalate the missing decisions. Exit 0 alone is not permission to execute every candidate action. + +Persona/live commands have command-specific verdicts. In particular, live exit 4 means a held writer lock, +a paused individual or a retired individual; do not retry it as though it were a transient engine error. Every recipe below was run in the cross-platform matrix (`.github/workflows/bench.yml`; transcripts in `scripts/bench/results/`), on the OSes named under each. @@ -55,12 +64,19 @@ node examples/node/decide.mjs router.json # decide your document ```js import { probbit } from './decide.mjs'; const r = await probbit(['decide', '--budget-ms', '200'], problem); // problem: an object or JSON text -if (r.kind === 'answer') act(r.output.plan, r.output.released); else escalate(r); +if (r.kind === 'answer' && r.output.plan_status === 'released') act(r.output.released_plan); +else escalate(r); // partial decisions need application-specific handling of coupled actions ``` The npm package (`npm/`) installs the binary and a `probbit` command with exit codes passed through. On Windows, spawn `probbit.exe` itself (set `PROBBIT_BIN`): Node cannot spawn npm's `probbit.cmd` shim without a shell. +## A host-enforced agent regression + +[examples/agent-harness](../examples/agent-harness/README.md) runs a local failing tool behind a real dispatch gate, +records its incident, reproduces a history-dependent retry change, then checks a repaired policy and replays its strand. +It requires no model or credentials. This is a bounded integration example, not a production permissions framework. + ## PowerShell Run on Windows (`windows-latest`) with PowerShell 7.6 as written. diff --git a/docs/persona.md b/docs/persona.md index 1ba387b..f5779df 100644 --- a/docs/persona.md +++ b/docs/persona.md @@ -640,6 +640,18 @@ ones it leaves unknown; the lint document gets `props` (each rule's `prove` entr Exit codes: `fuzz` 0 nothing found, 1 a counterexample; `prove` 0 every rule held or proved, 1 some rule unknown; both 2 bad input. JSON: `--json` (`probbit_persona_fuzz: 1`, `probbit_persona_prove: 1`); timing goes to stderr. +**Source-restricted fixtures.** For a persona with `reward_from`, exported abstract counterexamples have no runnable +`replay` or `explain` command unless a synthetic source is explicitly authorized with +`--fixture-src env:synthetic` (or `human:synthetic`, where allowed). The exporter strictly replays that labelled script; +it rejects incompatible source rules, records `fixture_provenance`, and never bypasses the production source check. +The Python `persona_fuzz` and MCP fuzz tool accept `fixture_src` too. See the +[host regression example](../examples/agent-harness/README.md). Synthetic observations are not production evidence. + +**Output formatting.** Single-document persona commands accept `--pretty` for stdout: init, turn, diff, lint, check, +compile, describe, fuzz and prove (for fuzz/prove it implies `--json`). Written states and programs remain canonical; +replay remains JSONL and explain remains text, so those commands reject `--pretty` with guidance. The +`probbit_persona_turn` field is the document schema version, not its zero-based `turn` counter. + ### 5.7 Live: a resident individual `probbit live` keeps one individual running: JSONL events in (one object of inputs per line, from `--events FILE` or stdin), @@ -770,7 +782,11 @@ against the replay (its `"checkpoints": N` is in the summary); `verify --from-ch checkpoint against the event line before it (prev, the stance digest, the state digest, the state read as any state is) and replays only what follows (`"from_checkpoint": n`); the lines before it are trusted, so run a full `verify` to check them. `monitor` starts at the last checkpoint and draws its event at once. A strand that ends in a checkpoint line is continued from -the state it carries. Measured on an Apple M4: a 10,000-event strand of a drives persona (4 goals, a floor, learning; 3,126,444 +the state it carries. Before continuation or a control append, the stored chain and lifecycle are checked from the header: +a checkpoint cannot hide a retired status, an invalid transition or a broken preceding hash link. This structural scan +costs time proportional to the existing log; it does not recompute earlier inference. Full `verify` is still necessary to +check that inference. Hashes alone do not authenticate a writer or detect a wholly rewritten log without a trusted external +anchor. Historical measurement before this hardening, on an Apple M4: a 10,000-event strand of a drives persona (4 goals, a floor, learning; 3,126,444 bytes, 10 checkpoint lines) opens in `monitor --once` from its last checkpoint in 0.005 s (load 3.3), where a full `verify` takes 192 s. A replay from a checkpoint costs what the events after it cost: 500 events past the last one took 26 s to draw (load 6-13, about 50 ms an event for this persona), so K bounds the wait; a smaller K costs a few KB per checkpoint line. A strand without checkpoint or control lines (shorter than K events, or `--checkpoint-every 0`) is the @@ -793,6 +809,7 @@ or names the line that differs. probbit monitor pip.strand --follow # bars in the terminal, redrawn as the strand grows probbit monitor pip.strand --open # the same board as a page at http://127.0.0.1:PORT/, in the browser probbit monitor --demo --open # the tutor's week, paced, for someone without a strand yet +probbit monitor --demo drives --open # eight synthetic events with goals and drives ``` **What it shows.** The latest event. A header: persona and version, seed, the individual's digest, the event number, the hours @@ -832,6 +849,9 @@ The page draws the terminal's board in a dark theme, the bars moving as the odds **Demo.** `probbit monitor --demo` replays the week of `probbit live examples/persona/tutor.yaml --seed 2 --demo week` (section 5.7; the week is written to a temporary file, read back and removed), paced 1 s per hour and each night in 2 s, under a minute: at a terminal, or with `--serve`, over and over. `--demo --once`, or a pipe, prints its last frame. +`--demo week` explicitly selects the same tutor demo; `--demo drives` selects the embedded Scout persona's eight-event +drive demo. Both are labelled synthetic, with a loop number and restart/completion cues. The default tutor events and +canonical replay documents are unchanged. Demo bars are scripted engine state, not measurements of an attached agent. Exit codes: 0 when every line replays, 1 at a line that differs (the frame names it and shows the event before it), 2 for a bad flag, a port that cannot be had, or a file that cannot be read or is not a strand. diff --git a/docs/probbit-ir-json.md b/docs/probbit-ir-json.md index 67630be..eec5180 100644 --- a/docs/probbit-ir-json.md +++ b/docs/probbit-ir-json.md @@ -137,7 +137,10 @@ returns `Err(String)` for structural errors (lengths, out-of-range indices, `i = weight difference beyond ~745 already decides (e^745 is past the range of double-precision probability ratios), so the limit removes no distribution you can express; it keeps every log-space quantity (log w sums over variables, pairs and same-group pairs; log Z) below ~1e9 x (input size), far inside the double range (1.8e308). Caps and limits are integers - 0..2^53. Above the limit a value is rejected, not clamped: rescale your scores. + 0..min(2^53, usize::MAX). On wasm32 the maximum is 4,294,967,295; size/count flags must also fit that target's + `usize`. The sum of weights in each linear cap must fit `usize` even when individual weights are valid. + Seeds remain 64-bit with the JSON exact-integer bound of 2^53. Out-of-range values are rejected, never wrapped + or silently clamped: rescale the model or use a wider native target. - **Arithmetic** stays in log space: Gibbs conditionals and enumeration subtract the running max before `exp`; the frontier DP carries log weights and merges by log-sum-exp (before 0.2.0 it used linear weights scaled per layer, which underflowed past ~745 nats; with weights of +-1000 it returned NaN marginals labelled `exact` and the CLI aborted, exit 134: probbit-ir test diff --git a/examples/agent-harness/README.md b/examples/agent-harness/README.md new file mode 100644 index 0000000..b36390a --- /dev/null +++ b/examples/agent-harness/README.md @@ -0,0 +1,72 @@ +# A recorded incident becomes an agent regression + +This small host runs a deliberately failing **local** probe. No model, credentials, network, +package install, or real external action is involved. Python 3.9+ and the built `probbit` +binary are the only requirements. + +From the repository root: + +```sh +cargo build --release -p probbit-cli +python3 examples/agent-harness/run.py --binary ./target/release/probbit +PROBBIT_BIN=./target/release/probbit python3 -m unittest discover -s examples/agent-harness -v +``` + +Use `--out /path/to/new-directory` to retain the strands, host action receipts, exact replay +copies, input events and proof document. Without it the demonstration uses temporary files. +On Windows, pass the path to `target/release/probbit.exe`; set `PROBBIT_BIN` using your shell's +environment syntax when running tests. + +## What runs + +1. The fresh `adaptive.json` policy dispatches two failing probes, then selects `ask`. +2. Twenty **explicitly synthetic** feedback events change its bounded learned weights. The same + failures now allow a third attempt. The independent host limit stops further dispatch. +3. `guarded.json` adds one hard habit: after two consecutive errors, select `ask`. The same + feedback and incident now dispatch only two calls; the host never executes a third. +4. Every incident replays into a separate strand with identical bytes and no tool calls. The + original incident is also replayed under the repaired policy as a regression. `persona prove` + checks the declared stance rule for seeds 0–19 and reports `held_by_construction`. + +The expected tool-call counts are **2 → 3 → 2**. These are a deterministic fixture, not a +production reliability or behavior-quality benchmark. Recorded durations vary between runs; +replay uses the actual captured elapsed-time inputs. + +## The host enforces the decision + +`run.py:gate` dispatches only the allowlisted `local_probe` when the stance is `ok`, the +`action` is released and equals `retry`, the habit violation count is zero, no inputs were +ignored, and the host attempt budget remains. Partial, refused, fallback, malformed and +unknown actions escalate without calling the tool. Engine errors abort before dispatch. +This is a real branch around a callable—not an instruction pasted into model prose. + +The host measures tool outcomes and elapsed time. `host-actions.json` correlates each proposal, +dispatch/stop, tool outcome and duration with the stance turn, state digest and strand head. +That action receipt is separate from the engine strand; the strand verifies decisions, not +external actions. There are no live approvals, distributed locks, crash-atomic dispatch, +exactly-once execution, or production authentication in this deliberately bounded example. +A plain circuit breaker can enforce this single static retry rule. The additional workflow +shown here is history-dependent reproduction, bounded learning, deterministic replay and a +scoped rule check. + +## Replaying generated counterexamples + +When `reward_from` requires a source, fuzz searches an abstract accepted-source history. +It must not silently label that history as a production observation. Explicitly authorize +a synthetic fixture source when exporting a runnable counterexample: + +```sh +./target/release/probbit persona fuzz examples/agent-harness/adaptive.json --seeds 0-19 --never '{when: {error_streak: 2}, then: {action: [ask]}}' --fixture-src env:synthetic --json +``` + +The command exits 1 when it finds a counterexample, 0 when none was found, and 2 on an input +error. `human:synthetic` is also accepted, but only where the persona permits human reward +sources. No source restriction is bypassed. The exported script and final stance are produced +by strict replay, and `fixture_provenance` records the synthetic assumption. Without the flag, +a source-required counterexample still reports the abstract script but its `replay` and +`explain` commands are `null`. Personas without `reward_from` keep their existing output. + +Do not copy synthetic labels onto production logs. In production the host must authenticate, +classify and record original observations. A source label or hash chain alone is not proof +of origin, and a stance proof says nothing about model obedience, subjective experience, or +whether the author's chosen rule covers every operational hazard. diff --git a/examples/agent-harness/adaptive.json b/examples/agent-harness/adaptive.json new file mode 100644 index 0000000..22c7194 --- /dev/null +++ b/examples/agent-harness/adaptive.json @@ -0,0 +1,60 @@ +{ + "probbit_persona": 1, + "identity": { + "name": "Retry policy", + "version": "1", + "seed": 0 + }, + "traits": [ + { + "id": "action", + "levels": [ + "retry", + "ask" + ], + "logw": [ + 0.5, + 0 + ], + "say": [ + "retry the local probe", + "escalate to the operator" + ] + } + ], + "inputs": [ + { + "id": "error", + "kind": "flag" + }, + { + "id": "praise", + "kind": "flag" + } + ], + "history": [ + { + "id": "error_streak", + "of": "error", + "kind": "streak", + "cap": 3, + "effects": { + "action": 0.2 + } + } + ], + "learning": { + "from": [ + "praise" + ], + "traits": [ + "action" + ], + "rate": 0.3, + "step_cap": 0.05, + "total_cap": 0.4 + }, + "reward_from": [ + "env" + ] +} diff --git a/examples/agent-harness/guarded.json b/examples/agent-harness/guarded.json new file mode 100644 index 0000000..542b6ea --- /dev/null +++ b/examples/agent-harness/guarded.json @@ -0,0 +1,74 @@ +{ + "probbit_persona": 1, + "identity": { + "name": "Retry policy", + "version": "1", + "seed": 0 + }, + "traits": [ + { + "id": "action", + "levels": [ + "retry", + "ask" + ], + "logw": [ + 0.5, + 0 + ], + "say": [ + "retry the local probe", + "escalate to the operator" + ] + } + ], + "inputs": [ + { + "id": "error", + "kind": "flag" + }, + { + "id": "praise", + "kind": "flag" + } + ], + "history": [ + { + "id": "error_streak", + "of": "error", + "kind": "streak", + "cap": 3, + "effects": { + "action": 0.2 + } + } + ], + "learning": { + "from": [ + "praise" + ], + "traits": [ + "action" + ], + "rate": 0.3, + "step_cap": 0.05, + "total_cap": 0.4 + }, + "reward_from": [ + "env" + ], + "habits": [ + { + "id": "escalate_after_two_errors", + "when": { + "error_streak": 2 + }, + "then": { + "action": [ + "ask" + ] + }, + "say": "two failures: stop and escalate" + } + ] +} diff --git a/examples/agent-harness/run.py b/examples/agent-harness/run.py new file mode 100644 index 0000000..19d8119 --- /dev/null +++ b/examples/agent-harness/run.py @@ -0,0 +1,145 @@ +#!/usr/bin/env python3 +"""Local, dependency-free incident-to-regression example. No model or network calls.""" +import argparse +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import time + +HERE = Path(__file__).resolve().parent +sys.path.insert(0, str(HERE.parent.parent / "python")) +import probbit + +RULE = {"when": {"error_streak": 2}, "then": {"action": ["ask"]}} + + +class FixtureFailure(Exception): + pass + + +class LocalProbe: + """Deliberately fails, without any external side effect.""" + def __init__(self): + self.calls = 0 + + def __call__(self): + self.calls += 1 + raise FixtureFailure("deliberate local fixture failure") + + +def gate(stance, proposal, attempts, max_attempts=3): + """The host is authoritative: only a released, clean retry dispatches the allowlisted tool.""" + if proposal != "local_probe": + return "escalate", "tool is not allowlisted" + if attempts >= max_attempts: + return "escalate", "independent host attempt limit" + if not isinstance(stance, dict) or stance.get("status") != "ok": + return "escalate", "stance not ok" + if stance.get("ignored") or stance.get("escalate"): + return "escalate", "ignored input or engine escalation" + habits = stance.get("habits") + if not isinstance(habits, dict) or habits.get("violations") != 0: + return "escalate", "habit verification unavailable or violated" + traits = stance.get("stance") + action = traits.get("action") if isinstance(traits, dict) else None + if not isinstance(action, dict) or action.get("released") is not True: + return "escalate", "action not released" + if action.get("level") != "retry": + return "escalate", "policy selected escalation" + return "dispatch", "released retry" + + +def command(binary, *args): + process = subprocess.run([binary, *map(str, args)], capture_output=True, encoding="utf-8", timeout=30) + if process.returncode != 0: + raise RuntimeError("command failed: " + process.stdout + process.stderr) + return process.stdout + + +def incident(policy, directory, binary, feedback=0): + """Host observations become the strand inputs; model prose is neither consulted nor executed.""" + directory.mkdir(parents=True) + document = json.loads(policy.read_text(encoding="utf-8")) + strand = directory / "incident.strand" + result = probbit.live_event(document, event={}, seed=0, strand=strand, binary=binary) + for _ in range(feedback): + # A clearly labelled test assumption, not production ratings or model self-reward. + result = probbit.live_event(document, result["state"], {"praise": True, "src": "env:synthetic"}, strand=strand, binary=binary) + tool = LocalProbe() + actions = [] + while True: + decision, reason = gate(result["stance"], "local_probe", tool.calls) + record = {"proposal": "local_probe", "decision": decision, "reason": reason, + "stance_turn": result["stance"]["turn"], "state_digest": result["state"]["digest"], + "strand_head": result["strand"]["head"], "attempts_before": tool.calls} + actions.append(record) + if decision != "dispatch": + break + started = time.monotonic_ns() + try: + tool() + except FixtureFailure as error: + elapsed_ns = time.monotonic_ns() - started + record.update(outcome="failure", duration_ns=elapsed_ns, detail=str(error)) + result = probbit.live_event(document, result["state"], + {"error": True, "src": "env:local_probe", "elapsed_hours": elapsed_ns / 3.6e12}, strand=strand, binary=binary) + else: + raise AssertionError("the demonstration probe always fails") + (directory / "host-actions.json").write_text(json.dumps(actions, indent=2) + "\n", encoding="utf-8") + verified = probbit.live_verify(strand, binary=binary) + if not verified["ok"]: + raise AssertionError(verified) + # Replay exactly the recorded inputs into a separate strand, with no tool calls at all. + records = [json.loads(line) for line in strand.read_text(encoding="utf-8").splitlines()] + events = [record["inputs"] for record in records[1:] if "n" in record] + event_file = directory / "events.jsonl" + event_file.write_text("".join(json.dumps(event) + "\n" for event in events), encoding="utf-8") + copy = directory / "replayed.strand" + command(binary, "live", policy, "--seed", "0", "--clock", "fixed", "--events", event_file, "--strand", copy) + if copy.read_bytes() != strand.read_bytes(): + raise AssertionError("incident replay changed strand bytes") + return {"tool_calls": tool.calls, "stop": actions[-1]["reason"], "replay_identical": True, + "events": verified["events"], "final_action": result["stance"]["stance"]["action"]["level"]}, events + + +def demonstrate(output, binary=None): + binary = probbit.find_binary(binary) + fresh, _ = incident(HERE / "adaptive.json", output / "fresh", binary) + learned, events = incident(HERE / "adaptive.json", output / "learned", binary, feedback=20) + guarded, _ = incident(HERE / "guarded.json", output / "guarded", binary, feedback=20) + replay = probbit.persona_replay(HERE / "guarded.json", events, seed=0, binary=binary) + # Events include actual tool failures. The same history becomes a checked regression. + failures = [stance for stance in replay if stance["inputs"].get("error")] + if failures[1]["stance"]["action"]["level"] != "ask": + raise AssertionError("the incident regression was not repaired") + proof = probbit.persona_prove(HERE / "guarded.json", never=RULE, seeds="0-19", threads=1, binary=binary) + if proof["properties"][0]["verdict"] != "held_by_construction": + raise AssertionError(proof) + if (fresh["tool_calls"], learned["tool_calls"], guarded["tool_calls"]) != (2, 3, 2): + raise AssertionError((fresh, learned, guarded)) + report = {"fresh": fresh, "after_synthetic_feedback": learned, "guarded_after_feedback": guarded, + "incident_regression": "passed", "stance_rule": proof["properties"][0]["verdict"], + "scope": "declared stance rule plus this local host gate; not model obedience, source authentication, or general agent safety"} + (output / "report.json").write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + (output / "proof.json").write_text(json.dumps(proof, indent=2) + "\n", encoding="utf-8") + return report + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--binary", help="probbit executable (otherwise PROBBIT_BIN/PATH/build fallback)") + parser.add_argument("--out", type=Path, help="new directory for the incident, actions and replay artifacts") + args = parser.parse_args() + if args.out: + args.out.mkdir(parents=True, exist_ok=False) + report = demonstrate(args.out, args.binary) + else: + with tempfile.TemporaryDirectory(prefix="probbit-agent-example-") as directory: + report = demonstrate(Path(directory), args.binary) + print(json.dumps(report, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/examples/agent-harness/test_harness.py b/examples/agent-harness/test_harness.py new file mode 100644 index 0000000..136239e --- /dev/null +++ b/examples/agent-harness/test_harness.py @@ -0,0 +1,38 @@ +"""Run: python3 -m unittest discover -s examples/agent-harness -v.""" +import copy +from pathlib import Path +import tempfile +import unittest +from run import demonstrate, gate + + +class HostEnforcement(unittest.TestCase): + def test_incident_replay_and_retry_repair(self): + with tempfile.TemporaryDirectory() as directory: + report = demonstrate(Path(directory)) + self.assertEqual(report["fresh"]["tool_calls"], 2) + self.assertEqual(report["after_synthetic_feedback"]["tool_calls"], 3) + self.assertEqual(report["guarded_after_feedback"]["tool_calls"], 2) + self.assertEqual(report["incident_regression"], "passed") + self.assertTrue(all(report[key]["replay_identical"] for key in ("fresh", "after_synthetic_feedback", "guarded_after_feedback"))) + + def test_host_rejects_unreleased_refused_or_unknown_tool_actions(self): + stance = {"status": "ok", "ignored": [], "habits": {"violations": 0}, + "stance": {"action": {"level": "retry", "released": True}}} + self.assertEqual(gate(stance, "local_probe", 0)[0], "dispatch") + cases = [None, {}, dict(stance, status="partial"), dict(stance, status="refused"), + dict(stance, status="fallback"), dict(stance, ignored=["typo"]), + dict(stance, escalate="ask"), dict(stance, habits={"violations": 1}), + dict(stance, habits=None), dict(stance, stance=[]), dict(stance, stance={"action": None})] + for key, value in (("released", False), ("level", "ask"), ("level", "arbitrary_command")): + changed = copy.deepcopy(stance) + changed["stance"]["action"][key] = value + cases.append(changed) + for result in cases: + self.assertEqual(gate(result, "local_probe", 0)[0], "escalate", result) + self.assertEqual(gate(stance, "shell", 0)[0], "escalate") + self.assertEqual(gate(stance, "local_probe", 3)[0], "escalate") + + +if __name__ == "__main__": + unittest.main() diff --git a/examples/node/decide.mjs b/examples/node/decide.mjs index 600cc9d..89873d5 100644 --- a/examples/node/decide.mjs +++ b/examples/node/decide.mjs @@ -11,7 +11,7 @@ const KIND = { 0: 'answer', 1: 'infeasible', 2: 'bad_input', 3: 'refused' }; /** * Run `probbit ` with `input` (an object or JSON text; omit for none) on stdin. Resolves to {kind, exitCode, output, stderr}: - * 'answer' exit 0: verdict exact | diagnostics_passed | partial (act on output.released, escalate output.escalated) + * 'answer' exit 0: verdict exact | diagnostics_passed | partial (inspect released_plan and escalate escalated) * 'infeasible' exit 1: no plan satisfies the rules (a proof, not an error) * 'bad_input' exit 2: output.error = {code, path, message} for a bad document; a bad flag leaves only a stderr line * 'refused' exit 3: verdict refused / declined (escalate), or output.error.code === 'numeric' @@ -47,9 +47,9 @@ export function probbit(args, input, { bin = process.env.PROBBIT_BIN || 'probbit function describe(r) { const o = r.output || {}; if (r.kind === 'bad_input') return o.error ? `${o.error.code} at ${o.error.path}: ${o.error.message}` : r.stderr; - const plan = o.plan ? Object.entries(o.plan).slice(0, 3).map(([t, w]) => `${t}->${w}`).join(', ') : ''; + const plan = o.released_plan ? Object.entries(o.released_plan).slice(0, 3).map(([t, w]) => `${t}->${w}`).join(', ') : ''; const released = Array.isArray(o.released) ? `, ${o.released.length} of ${o.tasks} released` : ''; - return `verdict ${o.verdict}${released}${plan ? `, plan ${plan}, ...` : ''}${o.reason ? ` (${o.reason})` : ''}`; + return `verdict ${o.verdict}${released}${plan ? `, released assignments ${plan}, ...` : ''}${o.plan_status === 'diagnostic' ? ', candidate is diagnostic only' : ''}${o.reason ? ` (${o.reason})` : ''}`; } async function main() { @@ -84,4 +84,4 @@ async function main() { process.exit(ok ? 0 : 1); } -if (import.meta.url === pathToFileURL(process.argv[1]).href) main(); +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) main(); diff --git a/install.ps1 b/install.ps1 index afc5c53..1d34815 100644 --- a/install.ps1 +++ b/install.ps1 @@ -42,6 +42,8 @@ if (-not $Version) { if (-not $Version) { throw 'probbit install: no published release found (set -Version)' } } if (-not $Version.StartsWith('v')) { $Version = "v$Version" } +if ($Version -notmatch '^v[A-Za-z0-9][A-Za-z0-9._+-]*$') { throw "probbit install: invalid release version: $Version" } +if ($Target -notmatch '^[A-Za-z0-9_-]+$') { throw "probbit install: invalid target: $Target" } if (-not $InstallDir) { $InstallDir = Join-Path $HOME '.local\bin' } $name = "probbit-$Version-$Target" @@ -64,11 +66,20 @@ try { Expand-Archive -LiteralPath $zip -DestinationPath (Join-Path $tmp 'x') -Force $src = Join-Path $tmp "x\$name\probbit.exe" if (-not (Test-Path -LiteralPath $src)) { throw "probbit install: $asset has no $name\probbit.exe" } + # Validate before touching an existing install. A wrong-platform or damaged executable must not replace it. + $ran = & $src version + if ($LASTEXITCODE -ne 0) { throw "probbit install: downloaded binary does not run here; existing install unchanged: $ran" } New-Item -ItemType Directory -Force -Path $InstallDir | Out-Null $dest = Join-Path $InstallDir 'probbit.exe' - Copy-Item -LiteralPath $src -Destination $dest -Force - $ran = & $dest version - if ($LASTEXITCODE -ne 0) { throw "probbit install: installed $dest, but it does not run here: $ran" } + $candidate = Join-Path $InstallDir ('.probbit-' + [guid]::NewGuid().ToString('N') + '.exe') + try { + Copy-Item -LiteralPath $src -Destination $candidate + # PowerShell coerces $null to an empty string for a string parameter; File.Replace rejects that path. + if (Test-Path -LiteralPath $dest) { [IO.File]::Replace($candidate, $dest, [NullString]::Value) } + else { [IO.File]::Move($candidate, $dest) } + } finally { + Remove-Item -LiteralPath $candidate -Force -ErrorAction SilentlyContinue + } Write-Host " installed $dest ($ran)" } finally { Remove-Item -LiteralPath $tmp -Recurse -Force -ErrorAction SilentlyContinue diff --git a/install.sh b/install.sh index 8c3e8f2..df2bb36 100755 --- a/install.sh +++ b/install.sh @@ -6,7 +6,7 @@ # # Environment (all optional): # PROBBIT_VERSION release tag, e.g. v0.8.0 (default: the latest release) -# PROBBIT_INSTALL_DIR where `probbit` goes (default: /usr/local/bin if you can write there, else ~/.local/bin) +# PROBBIT_INSTALL_DIR where `probbit` goes (default: ~/.local/bin; never overwrites a system-wide install implicitly) # PROBBIT_DOWNLOAD_BASE the archive is fetched from $PROBBIT_DOWNLOAD_BASE//probbit--.tar.gz # (default: https://github.com/BitmapAsset/probbit/releases/download); needs PROBBIT_VERSION # PROBBIT_TARGET Rust target triple to fetch instead of the detected one @@ -54,11 +54,15 @@ detect_target() { esac ;; Linux) - if (ldd --version 2>&1 || true) | grep -qi musl; then - die "this system uses musl libc (Alpine?) and the release ships a glibc build only; build from source: cargo install --git https://github.com/$REPO probbit-cli" - fi case "$arch" in - x86_64 | amd64) echo x86_64-unknown-linux-gnu ;; + x86_64 | amd64) + if (ldd --version 2>&1 || true) | grep -qi musl; then + echo x86_64-unknown-linux-musl + else + echo x86_64-unknown-linux-gnu + fi + ;; + aarch64 | arm64) echo aarch64-unknown-linux-musl ;; *) die "no prebuilt probbit for Linux on $arch yet; build from source: cargo install --git https://github.com/$REPO probbit-cli" ;; esac ;; @@ -91,6 +95,8 @@ main() { [ -n "$version" ] || die "no published release found (set PROBBIT_VERSION)" fi case "$version" in v*) ;; *) version="v$version" ;; esac + case "$version" in *[!A-Za-z0-9._+-]* | v) die "invalid release version: $version" ;; esac + case "$target" in *[!A-Za-z0-9_-]* | '') die "invalid target: $target" ;; esac name="probbit-$version-$target" asset="$name.tar.gz" @@ -109,11 +115,12 @@ main() { tar -xzf "$tmp/$asset" -C "$tmp" || die "could not unpack $asset" [ -f "$tmp/$name/probbit" ] || die "$asset has no $name/probbit" + # Check the candidate before replacing an existing working install (wrong architecture, corrupt executable, etc.). + chmod 755 "$tmp/$name/probbit" || die "could not make the downloaded binary executable" + ran=$("$tmp/$name/probbit" version 2>&1) || die "downloaded binary does not run here; existing install unchanged: $ran" if [ -n "${PROBBIT_INSTALL_DIR:-}" ]; then dir="$PROBBIT_INSTALL_DIR" - elif [ -d /usr/local/bin ] && [ -w /usr/local/bin ]; then - dir=/usr/local/bin else dir="${HOME:?HOME is not set; set PROBBIT_INSTALL_DIR}/.local/bin" fi @@ -121,7 +128,6 @@ main() { [ -w "$dir" ] || die "cannot write to $dir (set PROBBIT_INSTALL_DIR to a directory you can write; this script never uses sudo)" cp "$tmp/$name/probbit" "$dir/.probbit.$$" && chmod 755 "$dir/.probbit.$$" && mv -f "$dir/.probbit.$$" "$dir/probbit" \ || die "could not install into $dir" - ran=$("$dir/probbit" version 2>&1) || die "installed $dir/probbit, but it does not run here: $ran" say " installed $dir/probbit ($ran)" cmd=probbit @@ -130,7 +136,8 @@ main() { *) cmd="$dir/probbit" say "" - say "$dir is not on your PATH. Add it (e.g. in ~/.profile):" + case "${SHELL:-}" in */zsh) profile='~/.zshrc' ;; */bash) profile='~/.bashrc (or ~/.bash_profile for login shells)' ;; *) profile='your shell profile' ;; esac + say "$dir is not on your PATH. Add it in $profile:" say " export PATH=\"$dir:\$PATH\"" ;; esac diff --git a/npm/README.md b/npm/README.md index 28d62d6..af6c320 100644 --- a/npm/README.md +++ b/npm/README.md @@ -11,7 +11,7 @@ probbit stats --pretty ``` The install step fetches the release archive for your platform from GitHub Releases (macOS arm64 and x86_64, Linux -x86_64 with glibc, Windows x86_64), checks it against the `.sha256` file published beside it and keeps the binary inside +x86_64 with glibc or musl, Linux ARM64 static musl, Windows x86_64), checks it against the `.sha256` file published beside it and keeps the binary inside this package. No dependencies; Node >= 18. If the install step could not run (`--ignore-scripts`, offline), the first `probbit` call fetches the binary. A checksum mismatch fails the install and installs nothing. diff --git a/npm/bin/probbit.js b/npm/bin/probbit.js index 5e1d66e..b7da35d 100755 --- a/npm/bin/probbit.js +++ b/npm/bin/probbit.js @@ -16,15 +16,16 @@ async function main() { } const child = spawn(bin, process.argv.slice(2), { stdio: 'inherit' }); const signals = ['SIGINT', 'SIGTERM', 'SIGHUP'].filter((s) => process.platform !== 'win32' || s !== 'SIGHUP'); - const forward = (s) => { try { child.kill(s); } catch (e) { /* already gone */ } }; - for (const s of signals) process.on(s, forward); + // Node's signal events carry no argument. Bind each signal explicitly instead of accidentally forwarding SIGTERM. + const handlers = new Map(signals.map(s => [s, () => { try { child.kill(s); } catch (e) { /* already gone */ } }])); + for (const [s, handler] of handlers) process.on(s, handler); child.on('error', (e) => { process.stderr.write(`probbit: cannot run ${bin}: ${e.message}\n`); process.exit(127); }); child.on('exit', (code, signal) => { if (signal) { - for (const s of signals) process.removeListener(s, forward); + for (const [s, handler] of handlers) process.removeListener(s, handler); process.exitCode = 128 + (os.constants.signals[signal] || 0); // if the signal is ignored here (SIGPIPE) process.kill(process.pid, signal); // die the same way, so the shell sees 128 + n return; diff --git a/npm/install.js b/npm/install.js index 2d38ae4..42c5730 100644 --- a/npm/install.js +++ b/npm/install.js @@ -22,6 +22,7 @@ const TARGETS = { 'darwin-arm64': 'aarch64-apple-darwin', 'darwin-x64': 'x86_64-apple-darwin', 'linux-x64': 'x86_64-unknown-linux-gnu', + 'linux-arm64': 'aarch64-unknown-linux-musl', 'win32-x64': 'x86_64-pc-windows-msvc', 'win32-arm64': 'x86_64-pc-windows-msvc', // x64 emulation on Windows on Arm }; @@ -34,7 +35,7 @@ function binaryPath() { class ChecksumError extends Error {} async function get(url) { - const res = await fetch(url, { redirect: 'follow' }); + const res = await fetch(url, { redirect: 'follow', signal: AbortSignal.timeout(30000) }); if (!res.ok) throw new Error(`GET ${url}: HTTP ${res.status}`); return Buffer.from(await res.arrayBuffer()); } @@ -49,7 +50,8 @@ function place(src, dest) { fs.renameSync(tmp, dest); } catch (e) { fs.rmSync(tmp, { force: true }); - if (!fs.existsSync(dest)) throw e; // another process put it there first + // Only forgive an actual identical concurrent install, not an old binary left behind by a permission/rename error. + if (!fs.existsSync(dest) || !fs.readFileSync(src).equals(fs.readFileSync(dest))) throw e; } } @@ -62,13 +64,18 @@ async function install(log = (m) => process.stderr.write(`${m}\n`)) { log(`probbit: using ${src} (PROBBIT_BINARY)`); return dest; } - const target = env.PROBBIT_TARGET || TARGETS[`${process.platform}-${process.arch}`]; + const glibc = process.platform === 'linux' && process.report && process.report.getReport().header.glibcVersionRuntime; + const detected = process.platform === 'linux' && process.arch === 'x64' && !glibc + ? 'x86_64-unknown-linux-musl' : TARGETS[`${process.platform}-${process.arch}`]; + const target = env.PROBBIT_TARGET || detected; if (!target) { throw new Error(`no prebuilt probbit for ${process.platform}-${process.arch}; build one (cargo build --release -p probbit-cli) ` + 'and reinstall with PROBBIT_BINARY=/path/to/probbit'); } let tag = env.PROBBIT_VERSION || `v${pkg.version}`; if (!tag.startsWith('v')) tag = `v${tag}`; + if (!/^v[A-Za-z0-9][A-Za-z0-9._+-]*$/.test(tag)) throw new Error(`invalid release version: ${tag}`); + if (!/^[A-Za-z0-9_-]+$/.test(target)) throw new Error(`invalid target: ${target}`); const base = (env.PROBBIT_DOWNLOAD_BASE || `https://github.com/${REPO}/releases/download`).replace(/\/+$/, ''); const zip = target.includes('windows'); const name = `probbit-${tag}-${target}`; diff --git a/npm/test-wrapper.cjs b/npm/test-wrapper.cjs new file mode 100644 index 0000000..52aaacb --- /dev/null +++ b/npm/test-wrapper.cjs @@ -0,0 +1,92 @@ +'use strict'; +// Dependency-free unit checks for platform routing and native-process contracts. +const { test } = require('node:test'); +const assert = require('node:assert/strict'); +const fs = require('node:fs'); +const path = require('node:path'); +const vm = require('node:vm'); +const { EventEmitter } = require('node:events'); + +function installer(platform, arch, glibc, env = {}) { + const urls = []; + const module = { exports: {} }; + const context = { + require, module, __dirname, Buffer, AbortSignal, + process: { platform, arch, env, report: { getReport: () => ({ header: { glibcVersionRuntime: glibc } }) } }, + fetch: async url => { urls.push(url); throw new Error('fixture: network disabled'); }, + }; + vm.runInNewContext(fs.readFileSync(path.join(__dirname, 'install.js'), 'utf8'), context); + return { ...module.exports, urls }; +} + +for (const [platform, arch, glibc, target] of [ + ['darwin', 'arm64', null, 'aarch64-apple-darwin'], + ['darwin', 'x64', null, 'x86_64-apple-darwin'], + ['linux', 'x64', '2.35', 'x86_64-unknown-linux-gnu'], + ['linux', 'x64', null, 'x86_64-unknown-linux-musl'], + ['linux', 'arm64', '2.35', 'aarch64-unknown-linux-musl'], + ['win32', 'x64', null, 'x86_64-pc-windows-msvc'], + ['win32', 'arm64', null, 'x86_64-pc-windows-msvc'], +]) { + test(`archive selection: ${platform}/${arch}/${glibc || 'no glibc'}`, async () => { + const fixture = installer(platform, arch, glibc); + await assert.rejects(fixture.install(() => {}), /network disabled/); + assert.ok(fixture.urls[0].includes(`-${target}.`), fixture.urls[0]); + assert.equal(fixture.urls.length, 2); + }); +} + +test('unsupported architecture fails without fetching', async () => { + const fixture = installer('linux', 'riscv64', null); + await assert.rejects(fixture.install(() => {}), /no prebuilt/); + assert.equal(fixture.urls.length, 0); +}); + +for (const env of [{ PROBBIT_VERSION: '../bad' }, { PROBBIT_TARGET: '../../bad' }]) { + test(`invalid download selector: ${Object.keys(env)[0]}`, async () => { + const fixture = installer('darwin', 'arm64', null, env); + await assert.rejects(fixture.install(() => {}), /invalid/); + assert.equal(fixture.urls.length, 0); + }); +} + +function wrapper(platform = 'linux') { + const child = new EventEmitter(); + const signals = []; + child.kill = signal => signals.push(signal); + const process = new EventEmitter(); + Object.assign(process, { env: { PROBBIT_BINARY: '/fixture/probbit' }, argv: ['node', 'wrapper', 'monitor', '--follow'], + platform, pid: 42, stderr: { write() {} }, exit(code) { this.exitCode = code; }, kill(pid, signal) { this.killed = [pid, signal]; } }); + let invocation; + const requireFixture = id => { + if (id === 'child_process') return { spawn: (binary, args, opts) => { invocation = { binary, args: [...args], stdio: opts.stdio }; return child; } }; + if (id === '../install.js') return { binaryPath: () => '/unused', install: () => { throw new Error('unexpected download'); } }; + return require(id); + }; + vm.runInNewContext(fs.readFileSync(path.join(__dirname, 'bin/probbit.js'), 'utf8'), { require: requireFixture, process }); + return { child, process, signals, invocation }; +} + +test('wrapper preserves arguments, streams, and every documented exit code', () => { + for (const code of [0, 1, 2, 3, 4]) { + const fixture = wrapper(); + assert.deepEqual(fixture.invocation, { binary: '/fixture/probbit', args: ['monitor', '--follow'], stdio: 'inherit' }); + fixture.child.emit('exit', code, null); + assert.equal(fixture.process.exitCode, code); + } +}); + +test('wrapper forwards the exact signal even though Node signal events carry no arguments', () => { + const fixture = wrapper(); + for (const signal of ['SIGINT', 'SIGTERM', 'SIGHUP']) fixture.process.emit(signal); + assert.deepEqual(fixture.signals, ['SIGINT', 'SIGTERM', 'SIGHUP']); + fixture.child.emit('exit', null, 'SIGINT'); + assert.deepEqual(fixture.process.killed, [42, 'SIGINT']); + assert.equal(fixture.process.listenerCount('SIGINT'), 0); +}); + +test('wrapper reports native spawn failures as 127', () => { + const fixture = wrapper(); + fixture.child.emit('error', new Error('ENOENT')); + assert.equal(fixture.process.exitCode, 127); +}); diff --git a/playground/check-browser.mjs b/playground/check-browser.mjs new file mode 100644 index 0000000..fbd53b9 --- /dev/null +++ b/playground/check-browser.mjs @@ -0,0 +1,139 @@ +// Real Chromium UI checks, with no npm dependencies. Node >=22 and Chrome/Chromium are test-only requirements. +// Build first: sh playground/build.sh. Then: node playground/check-browser.mjs [screenshot-directory] +import assert from 'node:assert/strict'; +import { spawn, spawnSync } from 'node:child_process'; +import { createServer } from 'node:http'; +import { existsSync, mkdtempSync, mkdirSync, readFileSync, writeFileSync, rmSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { dirname, extname, join, resolve, sep } from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { once } from 'node:events'; + +const root = resolve(dirname(fileURLToPath(import.meta.url)), '..'); +const profile = mkdtempSync(join(tmpdir(), 'probbit-browser-')); +const output = resolve(process.argv[2] || mkdtempSync(join(tmpdir(), 'probbit-browser-shots-'))); +mkdirSync(output, { recursive: true }); +const executable = process.env.CHROME_BIN || [ + '/Applications/Google Chrome.app/Contents/MacOS/Google Chrome', + '/usr/bin/google-chrome', '/usr/bin/chromium', '/usr/bin/chromium-browser', +].find(existsSync) || spawnSync('which', ['google-chrome'], { encoding: 'utf8' }).stdout.trim(); +assert.ok(executable, 'Set CHROME_BIN to an installed Chrome or Chromium executable'); +assert.equal(typeof WebSocket, 'function', 'This test requires Node >=22 (built-in WebSocket)'); + +const server = createServer((req, res) => { + const pathname = decodeURIComponent(new URL(req.url, 'http://localhost').pathname); + const file = resolve(root, '.' + pathname); + if (!file.startsWith(root + sep)) { res.writeHead(403); res.end(); return; } + try { + const body = readFileSync(file); + const type = { '.html': 'text/html', '.js': 'text/javascript', '.wasm': 'application/wasm', '.json': 'application/json' }[extname(file)] || 'application/octet-stream'; + res.writeHead(200, { 'Content-Type': type }); res.end(body); + } catch { res.writeHead(404); res.end('Not found'); } +}); +server.listen(0, '127.0.0.1'); +await once(server, 'listening'); +const base = `http://127.0.0.1:${server.address().port}`; +const chrome = spawn(executable, ['--headless=new', '--no-first-run', '--no-default-browser-check', + '--disable-background-networking', '--remote-debugging-port=0', `--user-data-dir=${profile}`, 'about:blank'], +{ stdio: ['ignore', 'ignore', 'pipe'] }); +let socket; +const pending = new Map(); +const errors = []; +let nextId = 0; +try { + const url = await new Promise((ok, fail) => { + const timer = setTimeout(() => fail(new Error('Chrome debugger did not start within 20 seconds')), 20000); + let stderr = ''; + chrome.on('error', e => { clearTimeout(timer); fail(e); }); + chrome.on('exit', code => { clearTimeout(timer); fail(new Error(`Chrome exited: ${code}\n${stderr}`)); }); + chrome.stderr.on('data', b => { stderr += b; const match = stderr.match(/DevTools listening on (ws:\/\/\S+)/); if (match) { clearTimeout(timer); ok(match[1]); } }); + }); + socket = new WebSocket(url); + await new Promise((ok, fail) => { socket.addEventListener('open', ok, { once: true }); socket.addEventListener('error', fail, { once: true }); }); + socket.addEventListener('message', event => { + const msg = JSON.parse(event.data); + if (msg.method === 'Runtime.exceptionThrown') errors.push(msg.params.exceptionDetails); + if (msg.id && pending.has(msg.id)) { + const { ok, fail, timer } = pending.get(msg.id); pending.delete(msg.id); clearTimeout(timer); + if (msg.error) fail(new Error(JSON.stringify(msg.error))); else ok(msg.result); + } + }); + function call(method, params = {}, sessionId) { + const id = ++nextId; + return new Promise((ok, fail) => { + const timer = setTimeout(() => { pending.delete(id); fail(new Error(`CDP timeout: ${method}`)); }, 30000); + pending.set(id, { ok, fail, timer }); + socket.send(JSON.stringify({ id, method, params, ...(sessionId ? { sessionId } : {}) })); + }); + } + const { targetId } = await call('Target.createTarget', { url: 'about:blank' }); + const { sessionId } = await call('Target.attachToTarget', { targetId, flatten: true }); + const page = (method, params) => call(method, params, sessionId); + await page('Runtime.enable'); + await page('Page.enable'); + async function evaluate(expression) { + const r = await page('Runtime.evaluate', { expression, returnByValue: true, awaitPromise: true }); + if (r.exceptionDetails) throw new Error(JSON.stringify(r.exceptionDetails)); + return r.result.value; + } + async function until(expression) { + const deadline = Date.now() + 25000; + while (Date.now() < deadline) { + if (await evaluate(expression)) return; + await new Promise(r => setTimeout(r, 100)); + } + throw new Error(`UI timeout: ${expression}`); + } + async function screenshot(name) { + const shot = await page('Page.captureScreenshot', { format: 'png', captureBeyondViewport: false }); + writeFileSync(join(output, name + '.png'), Buffer.from(shot.data, 'base64')); + } + async function viewport(width) { + await page('Emulation.setDeviceMetricsOverride', { width, height: 1000, deviceScaleFactor: 1, mobile: false }); + } + await viewport(1440); + await page('Page.navigate', { url: base + '/playground/index.html' }); + await until('document.title === "probbit playground: ready"'); + let result = await evaluate('JSON.parse(document.querySelector("#raw").textContent)'); + assert.equal(result.verdict, 'exact'); assert.equal(result.violations, 0); + assert.equal(result.answers.team.probbit.value, 'technical'); + assert.match(await evaluate('document.querySelector("#meet-status").textContent'), /turns/); + await screenshot('playground-desktop'); + // Drive the public controls, including an error followed by recovery. + await evaluate('document.querySelector("#input").value = "{bad"; document.querySelector("#go").click()'); + await until('!document.querySelector("#go").disabled && document.querySelector("#verdict").textContent.startsWith("ERROR")'); + await evaluate('document.querySelector("#ex-demo").click(); document.querySelector("#go").click()'); + await until('!document.querySelector("#go").disabled && JSON.parse(document.querySelector("#raw").textContent).tasks === 300'); + result = await evaluate('JSON.parse(document.querySelector("#raw").textContent)'); + assert.equal(result.verdict, 'diagnostics_passed'); assert.equal(result.violations, 0); + await viewport(390); + await screenshot('playground-mobile'); + const playgroundWidth = await evaluate('({page: document.documentElement.scrollWidth, viewport: innerWidth})'); + assert.ok(playgroundWidth.page <= playgroundWidth.viewport + 1, `Playground overflow: ${JSON.stringify(playgroundWidth)}`); + console.log(JSON.stringify({ check: 'playground browser controls, errors, recovery, personas, 390px layout', passed: true })); + + await viewport(1440); + await page('Page.navigate', { url: base + '/playground/puzzle.html' }); + await until('document.querySelector("#lineA")?.textContent.includes("Stance:")'); + await evaluate('document.querySelector("#reveal").click(); document.querySelector("#replay").click()'); + await until('document.querySelector("#replayOut").textContent.includes("same")'); + const replay = await evaluate('document.querySelector("#replayOut").textContent'); + assert.match(replay, /Both branches repeat byte for byte and equal the native CLI's pinned values/); + assert.equal(await evaluate('document.querySelectorAll("#replayOut .bad").length'), 0); + await screenshot('puzzle-desktop'); + await viewport(390); + await screenshot('puzzle-mobile'); + const puzzleWidth = await evaluate('({page: document.documentElement.scrollWidth, viewport: innerWidth})'); + assert.ok(puzzleWidth.page <= puzzleWidth.viewport + 1, `Puzzle overflow: ${JSON.stringify(puzzleWidth)}`); + assert.equal(errors.length, 0, JSON.stringify(errors)); + console.log(JSON.stringify({ check: 'puzzle browser reveal and replay, 390px layout, no uncaught JS errors', passed: true })); + console.log(JSON.stringify({ screenshots: output, browser: (await call('Browser.getVersion')).product })); +} finally { + for (const { timer } of pending.values()) clearTimeout(timer); + if (socket) socket.close(); + chrome.kill('SIGTERM'); + await new Promise(resolve => { if (chrome.exitCode !== null) return resolve(); const timer = setTimeout(() => { chrome.kill('SIGKILL'); resolve(); }, 3000); chrome.once('exit', () => { clearTimeout(timer); resolve(); }); }); + server.closeAllConnections(); + await new Promise(resolve => server.close(resolve)); + rmSync(profile, { recursive: true, force: true }); +} diff --git a/playground/index.html b/playground/index.html index 0a556ac..5d5ff3d 100644 --- a/playground/index.html +++ b/playground/index.html @@ -20,6 +20,9 @@ #verdict { margin: 10px 0; font-size: 18px; font-weight: 700; letter-spacing: 0.04em; } .exact, .diagnostics_passed { color: var(--green); } .partial { color: var(--yellow); } .refused, .declined, .infeasible, .error { color: var(--red); } table { border-collapse: collapse; font: 13px/1.4 ui-monospace, SFMono-Regular, Menlo, monospace; margin-top: 6px; } + .table-scroll { max-width: 100%; overflow-x: auto; } + .table-scroll:focus-visible { outline: 2px solid var(--cyan); outline-offset: 3px; } + #runs, #verdict, .note { overflow-wrap: anywhere; } th, td { text-align: left; padding: 4px 12px 4px 0; border-bottom: 1px solid var(--line); vertical-align: top; } th { color: var(--dim); font-weight: 500; } td.moved { color: var(--magenta); } .note { color: var(--dim); font-size: 13px; margin-top: 6px; } details { margin-top: 16px; } summary { cursor: pointer; color: var(--dim); } @@ -39,7 +42,7 @@

probbit playground: WebAssembly, one thread, no server
-
+
full answer (JSON)

Meet three individuals from one persona

@@ -52,7 +55,7 @@

Meet three individuals from one persona

seeds -
+
Each cell: the stance line (habits first, then traits away from this individual's resting level), then three traits' levels with their exact odds. Habits hold in every cell (violations 0 by construction); the three differ where their genes do.
diff --git a/playground/puzzle.html b/playground/puzzle.html index 2bc7f37..6fd322c 100644 --- a/playground/puzzle.html +++ b/playground/puzzle.html @@ -17,6 +17,7 @@ h1 small { display: block; font-size: 14px; font-weight: 400; color: var(--dim); letter-spacing: 0.02em; } h2 { margin: 30px 0 8px; font-size: 18px; } p { max-width: 860px; } + .panel, .muted, .line { overflow-wrap: anywhere; min-width: 0; } .sub { color: var(--dim); margin: 10px 0 16px; } .bar { display: flex; flex-wrap: wrap; gap: 8px; align-items: center; margin: 10px 0; } button, select, input { background: var(--panel); color: var(--text); border: 1px solid var(--line); border-radius: 6px; padding: 7px 12px; font: inherit; } @@ -84,6 +85,8 @@ .caption strong { color: var(--text); } .caption span { color: var(--dim); } @media (max-width: 560px) { body { font-size: 14px; } + table { table-layout: fixed; } + th, td { overflow-wrap: anywhere; } .cards { grid-template-columns: repeat(2, minmax(0, 1fr)); } .card { min-height: 60px; } .lv { font-size: 12.5px; } diff --git a/probbit-cli/src/fuzz.rs b/probbit-cli/src/fuzz.rs index d03a6ab..8c974c3 100644 --- a/probbit-cli/src/fuzz.rs +++ b/probbit-cli/src/fuzz.rs @@ -107,6 +107,44 @@ pub fn fuzz(p: &Persona, props: &[Prop], s: &Search, eng: SyncEngine) -> (Vec>], source: &str, eng: persona::Engine) -> Result<(), crate::json::InErr> { + if !matches!(source, "env:synthetic" | "human:synthetic") { + return Err(persona::perr("fixture_src", "env:synthetic | human:synthetic; an explicit test-source assumption, never production provenance")); + } + if !persona::has_reward_sources(p) { return Err(persona::perr("fixture_src", "requires a persona with reward_from")); } + for per in res { + for (seed, f) in seeds.iter().zip(per) { + let Some(f) = f else { continue }; + let mut script = f.script.clone(); + for event in &mut script { + if event.iter().any(|(k, _)| k == "src") { return Err(persona::perr("fixture_src", "will not replace an existing source")); } + event.push(("src".into(), Json::Str(source.into()))); + persona::check_provenance(p, &persona::event_json(event))?; + } + let events: Vec = script.iter().map(|e| persona::event_json(e)).collect(); + let (stances, _) = persona::replay(p, Some(*seed), &events, false, eng, false)?; + f.doc = stances.last().expect("a counterexample has an event").clone(); + f.script = script; + } + } + Ok(()) +} +fn replayable(p: &Persona, f: &Found) -> bool { + f.script.iter().all(|e| persona::check_provenance(p, &persona::event_json(e)).is_ok()) +} +fn fixture_info(p: &Persona, f: &Found) -> Vec<(String, Json)> { + if !persona::has_reward_sources(p) { return vec![]; } + let source = f.script.first().and_then(|e| e.iter().find(|(k, _)| k == "src")).map(|(_, v)| v.clone()).unwrap_or(Json::Null); + vec![("fixture_provenance".into(), Json::Obj(vec![ + ("source".into(), source.clone()), ("synthetic".into(), Json::Bool(true)), + ("replayable".into(), Json::Bool(replayable(p, f))), + ("assumption".into(), Json::Str(if source.is_null() { + "search assumes accepted sources; use --fixture-src env:synthetic or human:synthetic to authorize synthetic replay; production evidence must retain its original source" + } else { "caller-authorized synthetic fixture source, not authenticated production evidence" }.into()))]))] +} /// `f(0) .. f(n-1)` on up to `threads` threads (each takes the next index; 8 MiB stacks, as the engine's threads), in index order fn par(n: usize, threads: usize, f: impl Fn(usize) -> T + Sync) -> Vec { let next = AtomicUsize::new(0); let slots: Vec>> = (0..n).map(|_| Mutex::new(None)).collect(); @@ -187,9 +225,10 @@ pub fn ranges(seeds: &[u64]) -> String { out.join(",") } fn script_json(sc: &[Event]) -> Json { Json::Arr(sc.iter().map(|e| persona::event_json(e)).collect()) } -fn commands(path: &str, seed: u64, f: &Found) -> (String, String) { +fn commands(p: &Persona, path: &str, seed: u64, f: &Found) -> Option<(String, String)> { + if !replayable(p, f) { return None; } let sc = word(&persona::canon(&script_json(&f.script))); - (format!("probbit persona replay {} --seed {seed} --script {sc}", word(path)), format!("probbit persona explain {} --seed {seed} --script {sc} --turn {}", word(path), f.script.len() - 1)) + Some((format!("probbit persona replay {} --seed {seed} --script {sc}", word(path)), format!("probbit persona explain {} --seed {seed} --script {sc} --turn {}", word(path), f.script.len() - 1))) } fn lengths(per: &[Option]) -> Vec<(usize, usize)> { let mut by: Vec<(usize, usize)> = vec![]; @@ -208,10 +247,11 @@ pub fn doc(p: &Persona, path: &str, props: &[Prop], s: &Search, res: &[Vec = props.iter().zip(res).map(|(pr, per)| { let failing: Vec = s.seeds.iter().zip(per).filter(|(_, f)| f.is_some()).map(|(sd, _)| n(*sd as f64)).collect(); - let sh = shortest(s, per).map_or(Json::Null, |(sd, f)| { let (rp, ex) = commands(path, sd, f); - Json::Obj(vec![("seed".into(), n(sd as f64)), ("script".into(), script_json(&f.script)), ("turn".into(), n((f.script.len() - 1) as f64)), ("broken".into(), broken_json(&f.broken)), - ("stance".into(), f.doc.clone()), ("replay".into(), st(&rp)), ("explain".into(), st(&ex))]) }); - let cex: Vec = s.seeds.iter().zip(per).filter_map(|(sd, f)| f.as_ref().map(|f| Json::Obj(vec![("seed".into(), n(*sd as f64)), ("script".into(), script_json(&f.script))]))).collect(); + let sh = shortest(s, per).map_or(Json::Null, |(sd, f)| { let cmds = commands(p, path, sd, f); + let (rp, ex) = cmds.map_or((Json::Null, Json::Null), |(r, e)| (st(&r), st(&e))); + let mut fields = vec![("seed".into(), n(sd as f64)), ("script".into(), script_json(&f.script)), ("turn".into(), n((f.script.len() - 1) as f64)), ("broken".into(), broken_json(&f.broken)), + ("stance".into(), f.doc.clone()), ("replay".into(), rp), ("explain".into(), ex)]; fields.extend(fixture_info(p, f)); Json::Obj(fields) }); + let cex: Vec = s.seeds.iter().zip(per).filter_map(|(sd, f)| f.as_ref().map(|f| { let mut fields = vec![("seed".into(), n(*sd as f64)), ("script".into(), script_json(&f.script))]; fields.extend(fixture_info(p, f)); Json::Obj(fields) })).collect(); Json::Obj(vec![("id".into(), st(&pr.id)), ("rule".into(), pr.rule.clone()), ("verdict".into(), st(if failing.is_empty() { "none_found" } else { "counterexample" })), ("individuals".into(), n(s.seeds.len() as f64)), ("failing".into(), n(failing.len() as f64)), ("failing_seeds".into(), Json::Arr(failing)), ("by_length".into(), Json::Obj(lengths(per).into_iter().map(|(l, c)| (l.to_string(), n(c as f64))).collect())), ("shortest".into(), sh), ("counterexamples".into(), Json::Arr(cex))]) }).collect(); @@ -246,7 +286,9 @@ pub fn human(p: &Persona, path: &str, props: &[Prop], s: &Search, res: &[Vec = s.seeds.iter().zip(per).filter(|(_, f)| f.is_some()).map(|(x, _)| *x).collect(); o.push(format!(" failing seeds: {}", ranges(&seeds))); } } @@ -274,7 +316,7 @@ pub fn seeds(v: &str) -> Option> { pub fn tool(args: &[(String, Json)], eng: SyncEngine) -> Result { use persona::perr; let get = |k: &str| args.iter().find(|(x, _)| x == k).map(|(_, v)| v).filter(|v| !v.is_null()); - const KNOWN: [&str; 12] = ["persona", "persona_path", "never", "props", "seeds", "fuzz_seed", "scripts", "depth", "beam", "grid", "hours", "threads"]; + const KNOWN: [&str; 13] = ["persona", "persona_path", "never", "props", "seeds", "fuzz_seed", "scripts", "depth", "beam", "grid", "hours", "threads", "fixture_src"]; let mut extra: Vec<&str> = args.iter().map(|(k, _)| k.as_str()).filter(|k| !KNOWN.contains(k)).collect(); extra.sort_unstable(); if let Some(k) = extra.first() { return Err(perr(&format!("arguments.{k}"), "unknown argument")); } let (p, path) = match (get("persona"), get("persona_path")) { @@ -301,7 +343,11 @@ pub fn tool(args: &[(String, Json)], eng: SyncEngine) -> Result Json { persona::run_program(prog, f, 1, 100, 0) } + #[test] + fn synthetic_fixture_export_replays_strictly_without_weakening_sources() { + let p = persona::build(&json::parse(r#"{"probbit_persona":1,"identity":{"name":"Fixture","version":"1"}, + "traits":[{"id":"action","levels":["retry","ask"],"logw":[1,0]}], + "inputs":[{"id":"praise","kind":"flag"},{"id":"criticism","kind":"flag"}], + "learning":{"from":["praise","criticism"],"traits":["action"],"rate":3,"step_cap":2,"total_cap":2}, + "reward_from":["env"]}"#).unwrap()).unwrap(); + let prs = persona::props(&p, &json::parse(r#"{"then":{"action":["retry"]}}"#).unwrap(), "never", "never").unwrap(); + let s = Search { seeds: vec![0], fuzz_seed: 0, scripts: 0, depth: 3, beam: 4, grid: vec![0.0, 1.0], hours: vec![1.0], threads: 1 }; + let (mut res, turns) = fuzz(&p, &prs, &s, &run); + let before = doc(&p, "fixture.json", &prs, &s, &res, turns); + let shortest = before.get("properties").unwrap().as_arr().unwrap()[0].get("shortest").unwrap(); + assert_eq!(shortest.get("replay"), Some(&Json::Null)); + let events = shortest.get("script").unwrap().as_arr().unwrap(); + assert!(persona::replay(&p, Some(0), events, false, &run, false).is_err()); + for src in ["env:production", "human:synthetic", "self", "clock"] { + assert!(authorize_fixtures(&p, &s.seeds, &mut res, src, &run).is_err(), "{src}"); + } + authorize_fixtures(&p, &s.seeds, &mut res, "env:synthetic", &run).unwrap(); + let f = res[0][0].as_ref().unwrap(); + let events = script_json(&f.script); + let (replayed, _) = persona::replay(&p, Some(0), events.as_arr().unwrap(), false, &run, false).unwrap(); + assert_eq!(replayed.last(), Some(&f.doc)); + assert!(commands(&p, "fixture.json", 0, f).is_some()); + assert!(authorize_fixtures(&p, &s.seeds, &mut res, "env:synthetic", &run).is_err(), "existing labels are never replaced"); + let state = persona::init(&p, Some(0), true, &run); + for ev in [r#"{"praise":true}"#, r#"{"praise":true,"src":"self"}"#] { + assert!(persona::turn(&p, &state, &json::parse(ev).unwrap(), false, &run, false).is_err()); + } + } + /// A habit of a generated persona: id, `when`, `then` (JSON text) and priority struct H { id: String, when: String, then: String, priority: usize } /// A random persona: 2-3 traits (2-3 levels), a mood (inertia 0.3-0.8) coupled to each, flags f0 and f1, a level input lv (neg, diff --git a/probbit-cli/src/json.rs b/probbit-cli/src/json.rs index 021d0b9..9022ffc 100644 --- a/probbit-cli/src/json.rs +++ b/probbit-cli/src/json.rs @@ -63,7 +63,7 @@ pub const MAX_WEIGHT: f64 = 1e9; /// Above it: a `limit` error before anything is allocated. The largest documented program is 200 x 65,535 = 13.1 million. pub const MAX_DENSE: usize = 20_000_000; pub fn weight(j: &Json, path: &str) -> Result { - let x = number(j, path)?; if x.abs() > MAX_WEIGHT { return Err(limit(path, format!("{x:e} is beyond the weight limit |x| <= 1e9 (natural-log odds; rescale)"))); } Ok(x) + let x = number(j, path)?; if !x.is_finite() || x.abs() > MAX_WEIGHT { return Err(limit(path, format!("{x:e} is beyond the finite weight limit |x| <= 1e9 (natural-log odds; rescale)"))); } Ok(x) } /// Paths of the non-finite numbers in `j` (a decision may carry none; main.rs `finish`). pub fn non_finite(j: &Json, path: &str, out: &mut Vec) { @@ -71,9 +71,14 @@ pub fn non_finite(j: &Json, path: &str, out: &mut Vec) { Json::Obj(v) => for (k, x) in v { non_finite(x, &at(path, k), out) }, _ => {} } } /// A whole number in 0..=2^53 (exact in a double): caps and limits. -pub fn count(j: &Json, path: &str) -> Result { +pub fn count_u64(j: &Json, path: &str) -> Result { let x = number(j, path)?; if x < 0.0 || x.fract() != 0.0 { return Err(value(path, "must be a non-negative integer")); } - if x > 9_007_199_254_740_992.0 { return Err(limit(path, "must be at most 2^53")); } Ok(x as usize) + if x > 9_007_199_254_740_992.0 { return Err(limit(path, "must be at most 2^53")); } Ok(x as u64) +} +/// A count that also fits this platform (wasm32 must not silently saturate it). +pub fn count(j: &Json, path: &str) -> Result { + let x = count_u64(j, path)?; + usize::try_from(x).map_err(|_| limit(path, "integer exceeds this platform's size limit")) } /// An array of distinct strings (`values`, `allowed`, `forbid`, cap `vars`). pub fn names<'a>(j: &'a Json, path: &str) -> Result, InErr> { @@ -103,8 +108,24 @@ fn node(b: &[u8], i: &mut usize, d: usize) -> Result { b't' if b[*i..].starts_with(b"true") => { *i += 4; Ok(Json::Bool(true)) } b'f' if b[*i..].starts_with(b"false") => { *i += 5; Ok(Json::Bool(false)) } b'n' if b[*i..].starts_with(b"null") => { *i += 4; Ok(Json::Null) } - b'-' | b'0'..=b'9' => { let s = *i; *i += 1; - while *i < b.len() && matches!(b[*i], b'0'..=b'9' | b'.' | b'e' | b'E' | b'+' | b'-') { *i += 1; } + b'-' | b'0'..=b'9' => { let s = *i; + // JSON's number grammar is narrower than Rust's f64 parser (which + // also accepts 01, 1., and -.1). Require each digit group explicitly. + if b[*i] == b'-' { *i += 1; } + match b.get(*i) { + Some(b'0') => *i += 1, + Some(b'1'..=b'9') => { *i += 1; while b.get(*i).is_some_and(u8::is_ascii_digit) { *i += 1; } } + _ => return Err(format!("bad number at byte {s}: expected an integer part")), + } + if b.get(*i) == Some(&b'.') { *i += 1; let start = *i; + while b.get(*i).is_some_and(u8::is_ascii_digit) { *i += 1; } + if *i == start { return Err(format!("bad number at byte {s}: expected fractional digits")); } + } + if matches!(b.get(*i), Some(b'e' | b'E')) { *i += 1; + if matches!(b.get(*i), Some(b'+' | b'-')) { *i += 1; } let start = *i; + while b.get(*i).is_some_and(u8::is_ascii_digit) { *i += 1; } + if *i == start { return Err(format!("bad number at byte {s}: expected exponent digits")); } + } let x = std::str::from_utf8(&b[s..*i]).unwrap().parse::().map_err(|e| format!("bad number at byte {s}: {e}"))?; if !x.is_finite() { return Err(format!("limit: number {} at byte {s} is not a finite double", String::from_utf8_lossy(&b[s..*i]))); } Ok(Json::Num(x)) } c => Err(format!("unexpected character '{}' at byte {i}", c as char)), @@ -124,7 +145,9 @@ fn string(b: &[u8], i: &mut usize) -> Result { if (0xDC00..0xE000).contains(&lo) { cp = 0x10000 + ((cp - 0xD800) << 10) + (lo - 0xDC00); } else { *i = s; } } out.push(char::from_u32(cp).ok_or_else(|| format!("lone surrogate \\u{cp:04x} at byte {}", *i - 6))?); } _ => return Err(format!("bad escape at byte {i}")) } } - _ => { let s = *i; while *i < b.len() && b[*i] != b'"' && b[*i] != b'\\' { *i += 1; } out.push_str(std::str::from_utf8(&b[s..*i]).map_err(|e| e.to_string())?); } } } + _ => { let s = *i; while *i < b.len() && b[*i] != b'"' && b[*i] != b'\\' { + if b[*i] < 0x20 { return Err(format!("unescaped control character at byte {i}")); } *i += 1; + } out.push_str(std::str::from_utf8(&b[s..*i]).map_err(|e| e.to_string())?); } } } } /// The 4 hex digits of a `\u` escape at `*i` (advanced past them). diff --git a/probbit-cli/src/live.rs b/probbit-cli/src/live.rs index be04ea0..4506fed 100644 --- a/probbit-cli/src/live.rs +++ b/probbit-cli/src/live.rs @@ -131,7 +131,7 @@ pub fn utc_now() -> String { /// rules. `open` checks the header and rebuilds the individual; each `step` replays one event line on the fixed clock and /// checks it (`prev`, `n`, the stance and state digests, the bytes). After an error the replay is spent: the chain past the /// line that differs cannot be checked. -pub struct Replay { pub live: Live, pub header: Json, pub doc: Json, pub line: usize, pub from: Option, last: Option } +pub struct Replay { pub live: Live, pub header: Json, pub doc: Json, pub line: usize, pub from: Option, last: Option, checkpoint_ready: bool } impl Replay { /// The header line (without its line ending) -> the replay, ready for event lines; Err((1, what differs)) pub fn open(head: &str) -> Result { @@ -143,7 +143,7 @@ impl Replay { let st0 = State::read(&p, h.get("state").unwrap_or(&Json::Null)).map_err(|e| (1, format!("the initial state: {}: {}", e.path, e.msg)))?; let (live, header) = Live::start(p, &doc, st0, Clock::Fixed, h.get("engine").and_then(Json::as_str).unwrap_or("")); if header != head { return Err((1, "the header is not as written".into())); } - Ok(Replay { live, header: h, doc, line: 1, from: None, last: None }) + Ok(Replay { live, header: h, doc, line: 1, from: None, last: None, checkpoint_ready: false }) } /// The replay from a checkpoint line (`cp`, without its line ending, at 1-based line `at`; `before` = the event line before it) /// instead of the header: the header is read as by `open`; the checkpoint must follow `before` (prev), at its event count, with @@ -154,6 +154,12 @@ impl Replay { let j = json::parse(cp).map_err(|e| (at, format!("not JSON: {}", e.msg)))?; let b = json::parse(before).map_err(|e| (at - 1, format!("not JSON: {}", e.msg)))?; if persona::canon(&j) != cp { return Err((at, "not canonical JSON".into())); } + if !j.as_obj().is_some_and(|kv| kv.len() == 4 && kv.iter().all(|(k, _)| ["checkpoint", "prev", "stance", "state"].contains(&k.as_str()))) { + return Err((at, "checkpoint fields must be checkpoint, prev, stance and state".into())); + } + if !b.as_obj().is_some_and(|kv| kv.len() == 5 && kv.iter().all(|(k, _)| ["n", "inputs", "prev", "stance", "state"].contains(&k.as_str()))) { + return Err((at - 1, "a checkpoint must immediately follow an event line".into())); + } if j.get("prev").and_then(Json::as_str) != Some(persona::digest_of(before).as_str()) { return Err((at, "prev is not the sha256 of the line before".into())); } let n = j.get("checkpoint").and_then(Json::as_f64).filter(|x| *x >= 1.0 && x.fract() == 0.0).ok_or((at, "not a checkpoint line".to_string()))? as u64; if b.get("n").and_then(Json::as_f64) != Some(n as f64) { return Err((at, format!("the line before is not event {n}"))); } @@ -161,14 +167,16 @@ impl Replay { if b.get("stance").and_then(Json::as_str) != Some(persona::sha(&stance).as_str()) { return Err((at, "the checkpoint's stance is not the one event {n} logs".replace("{n}", &n.to_string()))); } let st = State::read(&r.live.p, j.get("state").unwrap_or(&Json::Null)).map_err(|e| (at, format!("the checkpoint's state: {}: {}", e.path, e.msg)))?; if b.get("state").and_then(Json::as_str) != Some(st.digest.as_str()) { return Err((at, format!("the checkpoint's state is not the one event {n} logs"))); } - r.live.st = st; r.live.n = n; r.live.prev = persona::digest_of(cp); r.live.checkpoints = 1; r.line = at; r.from = Some(n); r.last = Some(stance); + r.live.st = st; r.live.n = n; r.live.prev = persona::digest_of(cp); r.live.checkpoints = 1; r.line = at; r.from = Some(n); r.last = Some(stance); r.checkpoint_ready = false; Ok(r) } /// The stance document of the last event replayed (or of the checkpoint the replay started from) pub fn last_stance(&self) -> Option<&Json> { self.last.as_ref() } /// One line (without its line ending) -> the stance document an event line replays to (None for a control or checkpoint /// line); Err((its 1-based line number, what differs)). A control line must be a valid move of the status and the line its - /// fields give; a checkpoint line must carry the event count and the state the replay reached. + /// fields give; a checkpoint line must immediately follow an event while active and carry + /// the event count and the state the replay reached. The last stance remains available to + /// the monitor after controls/checkpoints, but that does not authorize another checkpoint. pub fn step(&mut self, l: &str, eng: Engine) -> Result, (usize, String)> { let (i, live) = (self.line + 1, &mut self.live); let j = json::parse(l).map_err(|e| (i, format!("not JSON: {}", e.msg)))?; @@ -178,12 +186,14 @@ impl Replay { Kind::Control => { let f = |k: &str| j.get(k).and_then(Json::as_str).unwrap_or(""); let line = live.control(f("control"), f("by"), f("reason"), f("at")).map_err(|e| (i, format!("control line: {}: {}", e.path, e.msg)))?; if line != l { return Err((i, "the control line differs".into())); } - self.line = i; return Ok(None) } + self.line = i; self.checkpoint_ready = false; return Ok(None) } Kind::Checkpoint => { + if live.status != Status::Active { return Err((i, format!("checkpoint while {}", live.status.name()))); } + if !self.checkpoint_ready { return Err((i, "a checkpoint must immediately follow an event line".into())); } if j.get("checkpoint").and_then(Json::as_f64) != Some(live.n as f64) { return Err((i, format!("the checkpoint is not at event {}", live.n))); } let Some(stance) = self.last.as_ref() else { return Err((i, "a checkpoint follows an event line".into())) }; if live.checkpoint(stance) != l { return Err((i, "the checkpoint's state or stance differs from the replay".into())); } - self.line = i; return Ok(None) } + self.line = i; self.checkpoint_ready = false; return Ok(None) } Kind::Event => {} } if j.get("n").and_then(Json::as_f64) != Some((live.n + 1) as f64) { return Err((i, format!("n is not {}", live.n + 1))); } let ev = j.get("inputs").ok_or((i, "no inputs".to_string()))?; @@ -191,7 +201,7 @@ impl Replay { if j.get("stance").and_then(Json::as_str) != Some(persona::sha(&stance).as_str()) { return Err((i, "the stance differs".into())); } if j.get("state").and_then(Json::as_str) != Some(live.st.digest.as_str()) { return Err((i, "the state differs".into())); } if line != l { return Err((i, "the line differs".into())); } - self.line = i; self.last = Some(stance.clone()); + self.line = i; self.last = Some(stance.clone()); self.checkpoint_ready = true; Ok(Some(stance)) } /// `verify`'s summary of the lines replayed so far @@ -239,13 +249,60 @@ pub fn verify_doc(text: &str, eng: Engine) -> Json { /// This binary's engine (the strand header's `engine`) pub fn engine() -> String { format!("probbit {}", env!("CARGO_PKG_VERSION")) } +/// Validate the stored chain and lifecycle before extending it. This is deliberately not +/// inference replay or authentication: verify still recomputes every event. In particular, a +/// checkpoint cannot reactivate a retired individual or hide an invalid earlier control. +fn check_chain(text: &str) -> Result { + let ls = lines(text); + let err = |line: usize, why: String| perr("strand", format!("line {line}: {why}")); + let r = Replay::open(ls.first().copied().unwrap_or("")).map_err(|(i, why)| err(i, why))?; + let mut status = Status::Active; + let mut n = 0u64; + for i in 1..ls.len() { + let j = json::parse(ls[i]).map_err(|e| err(i + 1, e.msg))?; + if persona::canon(&j) != ls[i] { return Err(err(i + 1, "not canonical JSON".into())); } + if j.get("prev").and_then(Json::as_str) != Some(persona::digest_of(ls[i - 1]).as_str()) { + return Err(err(i + 1, "prev is not the sha256 of the line before".into())); + } + match kind(&j) { + Kind::Control => { + let f = |k: &str| j.get(k).and_then(Json::as_str).unwrap_or(""); + status = status.after(f("control")).map_err(|e| err(i + 1, e.msg))?; + let line = control_line(f("prev"), f("control"), f("by"), f("reason"), f("at")).map_err(|e| err(i + 1, e.msg))?; + if line != ls[i] { return Err(err(i + 1, "the control line differs".into())); } + } + Kind::Event => { + if status != Status::Active { return Err(err(i + 1, format!("event while {}", status.name()))); } + n += 1; + if j.get("n").and_then(Json::as_f64) != Some(n as f64) { return Err(err(i + 1, format!("n is not {n}"))); } + let Some(Json::Obj(_)) = j.get("inputs") else { return Err(err(i + 1, "no input object".into())) }; + for k in ["state", "stance"] { + if !j.get(k).and_then(Json::as_str).is_some_and(|s| s.len() == 71 && s.starts_with("sha256:") && s[7..].bytes().all(|c| c.is_ascii_hexdigit())) { + return Err(err(i + 1, format!("invalid {k} digest"))); + } + } + } + Kind::Checkpoint => { + if status != Status::Active { return Err(err(i + 1, format!("checkpoint while {}", status.name()))); } + // Full state validation plus links to the immediately preceding event; no inference. + let cp = Replay::open_at(ls[0], ls[i - 1], ls[i], i + 1).map_err(|(at, why)| err(at, why))?; + if cp.live.n != n { return Err(err(i + 1, "checkpoint event count differs".into())); } + // Keep the header persona/state check independent of the caller's state. + debug_assert_eq!(cp.live.p.digest, r.live.p.digest); + } + } + } + Ok(status) +} + /// Continue a strand: `text` is the strand so far and `st` the individual after its last line -> the individual, ready for the /// next event (no header to write). The header's persona must be `p` (the same digest); the run uses the header's document, the /// key order the strand replays with. The lines before are not replayed here (`verify` does that). pub fn resume(text: &str, p: &Persona, st: State, clock: Clock) -> Result { let bad = |m: String| perr("strand", m); if !text.ends_with('\n') { return Err(bad("its last line is incomplete (no newline at the end)".into())); } - let lines: Vec<&str> = text.split('\n').map(bare).filter(|l| !l.is_empty()).collect(); + check_chain(text)?; + let lines = lines(text); let h = json::parse(lines.first().copied().unwrap_or("")).map_err(|e| bad(format!("the header is not JSON: {}", e.msg)))?; if h.get("probbit_strand").and_then(Json::as_f64) != Some(FORMAT) { return Err(bad("not a probbit strand (format 1)".into())); } let q = persona::build(h.get("document").unwrap_or(&Json::Null)).map_err(|e| bad(format!("the header's persona: {}: {}", e.path, e.msg)))?; @@ -366,10 +423,7 @@ pub fn control_cmd(path: &str, what: &str, by: &str, reason: &str, at: &str) -> if !text.ends_with('\n') { return Err(perr("strand", "its last line is incomplete (no newline at the end)")); } let ls = lines(&text); if json::parse(ls.first().copied().unwrap_or("")).ok().and_then(|h| h.get("probbit_strand").and_then(Json::as_f64)) != Some(FORMAT) { return Err(perr("strand", "not a probbit strand (format 1)")); } - // the status: the control lines after the last event or checkpoint line (an event or a checkpoint is written only while active) - let mut status = Status::Active; - let tail: Vec = ls.iter().skip(1).rev().map_while(|l| json::parse(l).ok().filter(|j| matches!(kind(j), Kind::Control))).collect(); - for c in tail.iter().rev() { status = status.after(c.get("control").and_then(Json::as_str).unwrap_or("")).unwrap_or(Status::Retired); } + let status = check_chain(&text)?; let to = status.after(what)?; let line = control_line(&persona::digest_of(ls[ls.len() - 1]), what, by, reason, at)?; if !lock.held() { return Err(locked(None, &format!("{path}.lock"))); } @@ -623,6 +677,112 @@ mod tests { assert!(resume(&text, &p, again.st.clone(), Clock::Fixed).is_ok()); } + #[test] + fn a_checkpoint_cannot_smuggle_unknown_fields_into_fast_replay() { + let doc = json::parse(DOC).unwrap(); let p = persona::build(&doc).unwrap(); + let (mut lv, header) = Live::start(p.clone(), &doc, persona::init(&p, None, true, &run), Clock::Fixed, &engine()); + let (stance, event) = lv.event(&ev(0), &run).unwrap(); let cp = lv.checkpoint(&stance); + assert!(Replay::open_at(&header, &event, &cp, 3).is_ok()); + let mut extra = json::parse(&cp).unwrap(); + if let Json::Obj(kv) = &mut extra { kv.push(("extra".into(), Json::Bool(true))); } + assert!(Replay::open_at(&header, &event, &persona::canon(&extra), 3).is_err()); + } + + #[test] + fn resuming_or_controlling_a_damaged_chain_is_fail_closed() { + let (text, _) = strand(3); + let p = persona::build(&json::parse(DOC).unwrap()).unwrap(); + let mut replay = Replay::open(lines(&text)[0]).unwrap(); + for line in lines(&text).iter().skip(1) { replay.step(line, &run).unwrap(); } + let mut damaged = lines(&text).iter().map(|l| l.to_string()).collect::>(); + damaged[1] = damaged[1].replace("\"elapsed_hours\":", "\"unused_hours\":"); + let damaged = damaged.join("\n") + "\n"; + assert!(resume(&damaged, &p, replay.live.st.clone(), Clock::Fixed).is_err()); + assert!(resume(&text.replacen('\n', "\n\n", 1), &p, replay.live.st.clone(), Clock::Fixed).is_err()); + let f = std::env::temp_dir().join(format!("probbit-damaged-chain-{}.strand", std::process::id())); + std::fs::write(&f, &damaged).unwrap(); + assert!(control_cmd(f.to_str().unwrap(), "pause", "human:owner", "check", "test").is_err()); + assert_eq!(std::fs::read_to_string(&f).unwrap(), damaged); + assert!(!std::path::Path::new(&format!("{}.lock", f.display())).exists()); + std::fs::remove_file(f).unwrap(); + } + + /// A hash-consistent snapshot is not a legal checkpoint unless it immediately follows + /// an event. Keep full verification, fast verification and both writer paths aligned. + #[test] + fn illegal_checkpoint_order_is_refused_by_full_fast_append_and_control() { + let doc = json::parse(DOC).unwrap(); let p = persona::build(&doc).unwrap(); + for (name, controls) in [("paused", vec!["pause"]), ("retired", vec!["retire"]), + ("resumed", vec!["pause", "resume"]), ("consecutive", vec![])] { + let (mut lv, header) = Live::start(p.clone(), &doc, persona::init(&p, None, true, &run), Clock::Fixed, &engine()); + let (stance, event) = lv.event(&ev(0), &run).unwrap(); + let mut text = format!("{header}\n{event}\n"); + if controls.is_empty() { text += &lv.checkpoint(&stance); text.push('\n'); } + for control in controls { text += &lv.control(control, "human:owner", "test", "test").unwrap(); text.push('\n'); } + // Construct a snapshot with matching state, stance and hash chain, but at an + // illegal position. This is precisely what used to pass full verification. + let illegal = lv.checkpoint(&stance); text += &illegal; text.push('\n'); + let bad_line = lines(&text).len(); + let (line, why) = verify(&text, &run).expect_err(name); + assert_eq!(line, bad_line, "{name}: {why}"); + assert!(why.contains("checkpoint"), "{name}: {why}"); + assert!(verify_from_checkpoint(&text, &run).is_err(), "{name}: fast verify"); + assert!(resume(&text, &p, lv.st.clone(), Clock::Fixed).is_err(), "{name}: resume"); + let file = std::env::temp_dir().join(format!("probbit-checkpoint-order-{}-{name}.strand", std::process::id())); + std::fs::write(&file, &text).unwrap(); + let path = file.to_str().unwrap(); + let args = vec![("persona".into(), doc.clone()), ("state".into(), lv.st.to_json(&p)), + ("event".into(), Json::Obj(vec![])), ("strand_path".into(), Json::Str(path.into()))]; + assert!(tool("probbit_live_event", &args, &run).is_err(), "{name}: append"); + assert_eq!(std::fs::read_to_string(&file).unwrap(), text, "{name}: append writes nothing"); + assert!(control_cmd(path, "retire", "human:owner", "test", "test").is_err(), "{name}: control"); + assert_eq!(std::fs::read_to_string(&file).unwrap(), text, "{name}: control writes nothing"); + assert!(!std::path::Path::new(&format!("{path}.lock")).exists(), "{name}: lock released"); + std::fs::remove_file(file).unwrap(); + } + } + + #[test] + fn a_new_event_after_resume_authorizes_one_checkpoint_and_keeps_the_last_stance() { + let doc = json::parse(DOC).unwrap(); let p = persona::build(&doc).unwrap(); + let (mut lv, header) = Live::start(p.clone(), &doc, persona::init(&p, None, true, &run), Clock::Fixed, &engine()); + let (first, event) = lv.event(&ev(0), &run).unwrap(); + let pause = lv.control("pause", "human:owner", "test", "test").unwrap(); + let resume = lv.control("resume", "human:owner", "test", "test").unwrap(); + let mut replay = Replay::open(&header).unwrap(); + for line in [&event, &pause, &resume] { replay.step(line, &run).unwrap(); } + assert_eq!(replay.last_stance(), Some(&first), "monitor still has the last event's stance"); + let (second, event2) = lv.event(&ev(1), &run).unwrap(); + let checkpoint = lv.checkpoint(&second); + replay.step(&event2, &run).unwrap(); replay.step(&checkpoint, &run).unwrap(); + assert_eq!(replay.last_stance(), Some(&second)); + let text = format!("{header}\n{event}\n{pause}\n{resume}\n{event2}\n{checkpoint}\n"); + let full = verify(&text, &run).unwrap(); let fast = verify_from_checkpoint(&text, &run).unwrap(); + assert_eq!(full.get("final_state"), fast.get("final_state")); + assert_eq!(full.get("last_line"), fast.get("last_line")); + // Starting *at* a checkpoint must not authorize another checkpoint either. + let duplicate = lv.checkpoint(&second); + let mut fast = Replay::open_at(&header, &event2, &checkpoint, 6).unwrap(); + assert!(fast.step(&duplicate, &run).is_err()); + } + + #[test] + fn chain_checks_do_not_let_an_event_or_checkpoint_hide_retirement() { + let doc = json::parse(DOC).unwrap(); let p = persona::build(&doc).unwrap(); + let (mut lv, header) = Live::start(p.clone(), &doc, persona::init(&p, None, true, &run), Clock::Fixed, &engine()); + let (stance, event) = lv.event(&ev(0), &run).unwrap(); + let retired = lv.control("retire", "human:owner", "finished", "test").unwrap(); + let cp = lv.checkpoint(&stance); // low-level constructor: the writer must reject this invalid ordering + let bad = format!("{header}\n{event}\n{retired}\n{cp}\n"); + assert!(check_chain(&bad).unwrap_err().msg.contains("checkpoint while retired")); + lv.status = Status::Active; // simulate a hash-consistent but invalid stored lifecycle + let (_, illegal) = lv.event(&ev(1), &run).unwrap(); + let bad = format!("{header}\n{event}\n{retired}\n{cp}\n{illegal}\n"); + assert!(resume(&bad, &p, lv.st.clone(), Clock::Fixed).is_err()); + let valid = format!("{header}\n{event}\n{retired}\n"); + assert_eq!(check_chain(&valid).unwrap(), Status::Retired); + } + /// `probbit_live_event` logs each event to a strand file (created, then continued from the returned state) and the file is the /// strand one run writes; `probbit_live_verify` replays it, and names the line of a changed input as an answer #[test] diff --git a/probbit-cli/src/main.rs b/probbit-cli/src/main.rs index 5a0da7b..7a1b592 100644 --- a/probbit-cli/src/main.rs +++ b/probbit-cli/src/main.rs @@ -133,7 +133,7 @@ fn summary_doc(doc: &Json, bars: &[f64], cmd: &str) -> Json { for (k, x) in v { if after_verdict && k != "tier" && k != "reason" { out.push(("counts".into(), obj(std::mem::take(&mut counts)))); after_verdict = false; } match k.as_str() { - "plan" | "odds" | "marginals" | "release_reason" | "top_plans" | "escalated" => {} + "plan" | "released_plan" | "odds" | "marginals" | "release_reason" | "top_plans" | "escalated" => {} "released" => { if let Some(r) = &released { out.push(("worst_released".into(), worst(r))); } if let Some(e) = &escalated { out.push(("worst_escalated".into(), worst(e))); } } @@ -201,7 +201,7 @@ const FLAG_HELP: [(&str, &str); 29] = [ ("--op", "decide|exact|sample decide = exact tiers, then the sampler; exact / sample = only those (default decide)"), ("--deadline-ms", "N whole-call deadline (exact N/4, then sampler + polish); answer has `deadline` + `phases`"), ("--pretty", "indented JSON"), ("--progress", "[MS] one JSONL telemetry line on stderr every MS ms (default 100)"), - ("--tasks", "N tasks to generate (default 12)"), ("--hard", "tight quotas + strong affinity"), + ("--tasks", "N tasks to generate, 1..3333333 (default 12)"), ("--hard", "tight quotas + strong affinity"), ("--max-input-mb", "N largest stdin document read, MB (default 256; 0 = no limit)"), ("--summary", "compact answer: verdict, counts, gate, the 5 worst released and escalated items, telemetry (no per-item tables);\n with --pretty and a terminal on stderr, also a boxed summary there"), ("--top", "live monitor on stderr while it runs (tier, sweeps, updates/s, CPU, peak RSS, gate; 10 Hz), erased at exit;\n only with a terminal on stderr; stdout unchanged"), @@ -218,7 +218,7 @@ fn help(cmd: &str) -> Option { "demo" => ("Emit a synthetic agent-routing document (JSON) for `probbit decide`.", "probbit demo [flags] > router.json"), "ir" => ("Print a router document as probbit-ir v0 text (the hardware-facing form).", "probbit ir < router.json"), "stats" => ("The processor's spec sheet: machine, build, effective controls + their source, measured updates/s.", "probbit stats [flags]"), - "mcp" => ("A Model Context Protocol server on stdio (JSON-RPC 2.0, one message per line; logs on stderr). Tools probbit_decide,\n probbit_run, probbit_stats, probbit_demo, probbit_evaluate: the commands' own JSON in and out. Exits when stdin closes. docs/agents.md.", "probbit mcp"), + "mcp" => ("A Model Context Protocol server on stdio (JSON-RPC 2.0, one message per line; logs on stderr). Tools probbit_decide,\n probbit_run, probbit_stats, probbit_demo, probbit_evaluate, probbit_persona_init, probbit_persona_turn, probbit_persona_fuzz,\n probbit_live_event, probbit_live_verify: the commands' own JSON in and out. Exits when stdin closes. docs/agents.md.", "probbit mcp"), "persona" => (PERSONA_HELP, "probbit persona PERSONA [flags]"), "live" => (LIVE_HELP, "probbit live PERSONA [--seed N | --state FILE] [--strand FILE] [--events FILE [--watch]] [--clock real|fixed] [--checkpoint-every K] | probbit live PERSONA --demo week [--seed N] [--strand FILE] [--plain] | probbit live verify STRAND [--from-checkpoint] | probbit live control STRAND pause|resume|retire --by WHO --reason TEXT [--at TIME]"), "version" => ("Print the version.", "probbit version"), _ => return None }; @@ -258,12 +258,12 @@ fn cycles_arg(args: &[String]) -> bool { fn check_flags(args: &[String], values: &[&str], switches: &[&str]) { let mut i = 1; let mut seen: Vec<&str> = Vec::new(); while i < args.len() { let a = args[i].as_str(); - if values.contains(&a) { if i + 1 >= args.len() { fail(&format!("{a} needs a value")) } + if values.contains(&a) { if args.get(i + 1).map_or(true, |v| v.starts_with("--")) { fail(&format!("{a} needs a value")) } if seen.contains(&a) { fail(&format!("{a} given twice")) } seen.push(a); i += 2; } else if a == "--progress" && switches.contains(&a) { if seen.contains(&a) { fail(&format!("{a} given twice")) } seen.push(a); i += if args.get(i + 1).is_some_and(|v| v.parse::().is_ok()) { 2 } else { 1 }; } else if switches.contains(&a) { i += 1; } - else { fail(&format!("unknown argument {a:?} for `probbit {}` (run `probbit` for usage)", args[0])) } } + else { fail(&format!("unknown argument {a:?} for `probbit {}` (see `probbit {} --help`)", args[0], args[0])) } } } /// Optional config file: `$PROBBIT_CONFIG`, else `./probbit.json` if present (keys chains, threads, cpu_limit, mem_limit_mb, priority). fn config_path() -> Option { std::env::var("PROBBIT_CONFIG").ok().or_else(|| std::path::Path::new("probbit.json").exists().then(|| "probbit.json".to_string())) } @@ -436,8 +436,11 @@ fn demo_cmd(args: &[String]) { let (v, w) = flags_of("demo"); check_flags(args, &v, &w); let seed: u64 = arg(args, "--seed", 7); let hard = args.iter().any(|a| a == "--hard"); let live = args.iter().any(|a| a == "--live"); let th = theme::stderr(args); let auto = !live && th.is_some() && theme::stdout_is_terminal(); - if !(live || auto) { emit(&json::write(&demo_doc(arg(args, "--tasks", 12), seed, hard), true)); return; } - let doc = demo_doc(arg(args, "--tasks", 300), seed, hard); + let tasks: usize = arg(args, "--tasks", if live || auto { 300 } else { 12 }); + let max_tasks = json::MAX_DENSE / 6; // the generated router has six workers + if tasks == 0 || tasks > max_tasks { fail(&format!("--tasks must be 1 to {max_tasks} (the router's task-worker limit)")); } + let doc = demo_doc(tasks, seed, hard); + if !(live || auto) { emit(&json::write(&doc, true)); return; } let Some(th) = th else { err_line("probbit: --live draws on a colour terminal (stderr); no terminal, or NO_COLOR / --plain / PROBBIT_THEME=plain: printed the problem only"); emit(&json::write(&doc, true)); return }; let mut n = from_json(&doc).unwrap_or_else(|e| bad_input("bad problem: ", e)); n.p.collective = true; n.p.cluster = true; n.p.cycles = true; // decide's defaults @@ -472,7 +475,7 @@ fn evaluate_cmd(args: &[String]) { let c = finish(evaluate::respond(&r, doc), code, &View::new(args, "evaluate"), &bars); if c != 0 { std::process::exit(c); } } -const PERSONA_HELP: &str = "The individuality layer (docs/persona.md): a persona file (YAML subset or JSON: traits with priors, moods with\n inertia, couplings, per-turn inputs, history, habits = hard rules) + an individual's state + this turn's inputs -> ONE probbit-ir\n program, run in process (`probbit run --op decide` at fixed work) -> the stance: a level per trait with exact odds, the habits\n in force and the ones that bound, a refusal when the engine cannot vouch, a why and a short stance line for any model's prompt.\n Documents are canonical JSON (keys sorted); a turn is a pure function of (persona, state, inputs, version).\nsubcommands:\n init PERSONA [--seed N] [--out STATE] a new individual (genes from the seed, resting stance); stdout or --out\n turn PERSONA --state STATE [--inputs JSON|FILE] [--out STATE2] [--timing] [--no-inertia]\n the stance on stdout; the new state replaces STATE (or goes to --out)\n replay PERSONA [--seed N] --script JSON|FILE [--out TRACE] [--no-inertia] [--timing]\n init, then every turn of the script: one stance per line (JSONL)\n explain PERSONA [--seed N] --script JSON|FILE --turn K turn K in words: every contribution, the odds, the twin\n diff PERSONA [--seed A] [--other PERSONA2] [--seed2 B] --script JSON|FILE distance between two individuals\n lint PERSONA [--never RULE | --props FILE] [--seeds 0-99] [--threads N]\n contradicting habits (every conditional habit and pair, every prev level);\n warnings: planned levels that are not their trait's most likely one;\n with rules: prove each one, fuzz the unknown ones; exit 1 on a counterexample\n fuzz PERSONA (--never RULE | --props FILE) [--seeds 0-99] [--fuzz-seed N] [--scripts N] [--depth N] [--beam N]\n [--grid LIST] [--hours LIST] [--threads N] [--json]\n search event scripts for the shortest one whose stance breaks a rule\n (habit syntax: {when: {...}, then: {...}}); shrunk, replayable (§5.6)\n prove PERSONA (--never RULE | --props FILE) [--seeds 0-99] [--threads N] [--json]\n per rule: held by construction (a habit implies it), proved for every\n event sequence (a sound bound), or unknown (the cell it fails; fuzz it) (§5.6)\n check PERSONA valid? digest and sizes\n compile PERSONA --state STATE [--inputs JSON|FILE] the turn's probbit-ir program (its sha256 = engine.program)\n describe PERSONA traits, moods, inputs, habits, agenda\n A script is a JSON list of per-turn input objects, or {\"turns\": [...]}. Resource controls as for run (PROBBIT_THREADS, ...)."; +const PERSONA_HELP: &str = "The individuality layer (docs/persona.md): a persona file (YAML subset or JSON: traits with priors, moods with\n inertia, couplings, per-turn inputs, history, habits = hard rules) + an individual's state + this turn's inputs -> ONE probbit-ir\n program, run in process (`probbit run --op decide` at fixed work) -> the stance: a level per trait with exact odds, the habits\n in force and the ones that bound, a refusal when the engine cannot vouch, a why and a short stance line for any model's prompt.\n Documents are canonical JSON (keys sorted); a turn is a pure function of (persona, state, inputs, version).\n --pretty indents a single JSON document on stdout (init, turn, diff, lint, check, compile, describe, fuzz, prove);\n fuzz/prove --pretty implies --json. State files stay canonical. Replay stays JSONL; explain stays plain text.\nsubcommands:\n init PERSONA [--seed N] [--out STATE] a new individual (genes from the seed, resting stance); stdout or --out\n turn PERSONA --state STATE [--inputs JSON|FILE] [--out STATE2] [--timing] [--no-inertia]\n the stance on stdout; the new state replaces STATE (or goes to --out)\n replay PERSONA [--seed N] --script JSON|FILE [--out TRACE] [--no-inertia] [--timing]\n init, then every turn of the script: one stance per line (JSONL)\n explain PERSONA [--seed N] --script JSON|FILE --turn K turn K in words: every contribution, the odds, the twin\n diff PERSONA [--seed A] [--other PERSONA2] [--seed2 B] --script JSON|FILE distance between two individuals\n lint PERSONA [--never RULE | --props FILE] [--seeds 0-99] [--threads N]\n contradicting habits (every conditional habit and pair, every prev level);\n warnings: planned levels that are not their trait's most likely one;\n with rules: prove each one, fuzz the unknown ones; exit 1 on a counterexample\n fuzz PERSONA (--never RULE | --props FILE) [--seeds 0-99] [--fuzz-seed N] [--scripts N] [--depth N] [--beam N]\n [--grid LIST] [--hours LIST] [--threads N] [--json] [--fixture-src env:synthetic|human:synthetic]\n search event scripts for the shortest one whose stance breaks a rule\n (habit syntax: {when: {...}, then: {...}}); shrunk (§5.6). --fixture-src\n explicitly authorizes synthetic-source fixtures for strict replay; never\n use them to relabel production events. The persona must allow that source.\n prove PERSONA (--never RULE | --props FILE) [--seeds 0-99] [--threads N] [--json]\n per rule: held by construction (a habit implies it), proved for every\n event sequence (a sound bound), or unknown (the cell it fails; fuzz it) (§5.6)\n check PERSONA valid? digest and sizes\n compile PERSONA --state STATE [--inputs JSON|FILE] the turn's probbit-ir program (its sha256 = engine.program)\n describe PERSONA traits, moods, inputs, habits, agenda\n A script is a JSON list of per-turn input objects, or {\"turns\": [...]}. Resource controls as for run (PROBBIT_THREADS, ...)."; /// The persona engine of this process: `persona::run_program` with the resource controls `probbit run` would use (flag > env > /// config > default; resolved once), and --priority low applied if asked for. They never change an answer at fixed work. pub(crate) fn persona_engine() -> impl Fn(&Json, &persona::Flags) -> Json { @@ -485,6 +488,11 @@ fn json_arg(s: &str, what: &str) -> Json { let t = if std::path::Path::new(s).is_file() { std::fs::read_to_string(s).unwrap_or_else(|e| bad_input("persona: ", persona::perr(what, format!("cannot read {s}: {e}")))) } else { s.to_string() }; json::parse(t.strip_prefix('\u{feff}').unwrap_or(&t)).unwrap_or_else(|e| bad_input("persona: ", persona::perr(what, format!("not JSON: {}", e.msg)))) } +/// Human-readable JSON without changing the canonical values, sorted keys, digests or stored state bytes. +fn persona_text(doc: &Json, pretty: bool) -> String { + let canonical = persona::canon(doc); + if pretty { json::write(&json::parse(&canonical).expect("canonical persona JSON"), true) } else { canonical } +} /// Write a document: to `path` (write, then rename: a killed turn never leaves a half-written state), or to stdout fn put(path: Option<&str>, text: &str) { let Some(path) = path else { emit(text); return }; @@ -528,20 +536,24 @@ fn rules_arg(args: &[String], p: &persona::Persona) -> Vec { } fn fuzz_cmd(args: &[String], path: &str, p: &persona::Persona, eng: fuzz::SyncEngine) { let props = rules_arg(args, p); + let pretty = args.iter().any(|a| a == "--pretty"); let sub = args[1].as_str(); if props.is_empty() { fail(&format!("persona {sub}: give a rule: --never RULE (habit syntax) or --props FILE")) } let bounded = |f: &str, d: usize, lo: usize, hi: usize| -> usize { let x: usize = arg(args, f, d); if x < lo || x > hi { fail(&format!("{f} must be {lo} to {hi}")) } x }; let cores = std::thread::available_parallelism().map_or(1, |n| n.get()); if sub == "prove" { let (seeds, threads) = (seeds_arg(args), bounded("--threads", cores, 1, 1024)); let t0 = std::time::Instant::now(); let out = fuzz::prove(p, &props, &seeds, threads, eng); let secs = t0.elapsed().as_secs_f64(); - if args.iter().any(|a| a == "--json") { emit(&persona::canon(&fuzz::prove_doc(p, &props, &seeds, &out))) } else { emit(&fuzz::prove_human(p, path, &props, &seeds, &out)) } + if pretty || args.iter().any(|a| a == "--json") { emit(&persona_text(&fuzz::prove_doc(p, &props, &seeds, &out), pretty)) } else { emit(&fuzz::prove_human(p, path, &props, &seeds, &out)) } err_line(&format!("prove: {secs:.2} s")); if out.iter().any(|v| matches!(v, fuzz::Verdict::Unknown { .. })) { std::process::exit(1) } return; } let s = fuzz::Search { seeds: seeds_arg(args), fuzz_seed: arg(args, "--fuzz-seed", 0u64), scripts: bounded("--scripts", 60, 0, 1_000_000), depth: bounded("--depth", 8, 1, 64), beam: bounded("--beam", 4, 0, 64), grid: list_arg(args, "--grid", &[0.0, 0.5, 1.0], 0.0, 1.0), hours: list_arg(args, "--hours", &[1.0, 12.0, 48.0], 0.0, 1e6), threads: bounded("--threads", cores, 1, 1024) }; - let t0 = std::time::Instant::now(); let (res, turns) = fuzz::fuzz(p, &props, &s, eng); let secs = t0.elapsed().as_secs_f64(); - if args.iter().any(|a| a == "--json") { emit(&persona::canon(&fuzz::doc(p, path, &props, &s, &res, turns))) } else { emit(&fuzz::human(p, path, &props, &s, &res, turns)) } + let t0 = std::time::Instant::now(); let (mut res, turns) = fuzz::fuzz(p, &props, &s, eng); let secs = t0.elapsed().as_secs_f64(); + if let Some(i) = args.iter().position(|a| a == "--fixture-src") { + fuzz::authorize_fixtures(p, &s.seeds, &mut res, &args[i + 1], eng).unwrap_or_else(|e| bad_input("persona fuzz: ", e)); + } + if pretty || args.iter().any(|a| a == "--json") { emit(&persona_text(&fuzz::doc(p, path, &props, &s, &res, turns), pretty)) } else { emit(&fuzz::human(p, path, &props, &s, &res, turns)) } err_line(&format!("fuzz: {turns} turns in {secs:.2} s ({:.0} turns/s, {} search thread{})", turns as f64 / secs.max(1e-9), s.threads.min(s.seeds.len().max(1)), if s.threads.min(s.seeds.len().max(1)) == 1 { "" } else { "s" })); if res.iter().any(|per| per.iter().any(Option::is_some)) { std::process::exit(1) } } @@ -638,11 +650,15 @@ fn persona_cmd(args: &[String]) { "init" => (&["--seed", "--out"], &[]), "turn" => (&["--state", "--inputs", "--out"], &["--timing", "--no-inertia"]), "replay" => (&["--seed", "--script", "--out"], &["--no-inertia", "--timing"]), "explain" => (&["--seed", "--script", "--turn"], &[]), "diff" => (&["--seed", "--other", "--seed2", "--script"], &[]), "compile" => (&["--state", "--inputs"], &[]), "check" | "describe" => (&[], &[]), "lint" => (&["--never", "--props", "--seeds", "--threads"], &[]), - "fuzz" => (&["--seeds", "--never", "--props", "--fuzz-seed", "--scripts", "--depth", "--beam", "--grid", "--hours", "--threads"], &["--json"]), + "fuzz" => (&["--seeds", "--never", "--props", "--fuzz-seed", "--scripts", "--depth", "--beam", "--grid", "--hours", "--threads", "--fixture-src"], &["--json"]), "prove" => (&["--seeds", "--never", "--props", "--threads"], &["--json"]), _ => fail("persona: a subcommand: init, turn, replay, explain, diff, lint, fuzz, prove, check, compile or describe (probbit persona --help)") }; + let pretty = args.iter().any(|a| a == "--pretty"); + if pretty && sub == "replay" { fail("persona replay emits one canonical JSON object per line (JSONL); --pretty is for single-document commands such as persona turn") } + if pretty && sub == "explain" { fail("persona explain already emits readable text; --pretty is for JSON commands") } + let mut switches = sw.to_vec(); switches.push("--pretty"); let Some(path) = args.get(2).filter(|a| !a.starts_with("--")) else { fail(&format!("persona {sub}: the persona file comes first: probbit persona {sub} PERSONA [flags]")) }; - let mut a = vec![format!("persona {sub}")]; a.extend(args[3..].iter().cloned()); check_flags(&a, vals, sw); + let mut a = vec![format!("persona {sub}")]; a.extend(args[3..].iter().cloned()); check_flags(&a, vals, &switches); let need = |f: &str| -> String { args.iter().position(|x| x == f).and_then(|k| args.get(k + 1)).cloned().unwrap_or_else(|| fail(&format!("persona {sub}: {f} is required"))) }; let opt = |f: &str| -> Option { args.iter().position(|x| x == f).and_then(|k| args.get(k + 1)).cloned() }; let (p, _) = persona::load(path).unwrap_or_else(|e| bad_input("persona: ", e)); @@ -655,9 +671,11 @@ fn persona_cmd(args: &[String]) { let inputs = || opt("--inputs").map_or(Json::Obj(vec![]), |s| json_arg(&s, "inputs")); let turns = || persona::script(&json_arg(&need("--script"), "script")).unwrap_or_else(|e| bad_input("persona: ", e)); match sub { - "init" => { let st = persona::init(&p, seed_arg(args, "--seed"), true, eng); put(opt("--out").as_deref(), &persona::canon(&st.to_json(&p))); } + "init" => { let st = persona::init(&p, seed_arg(args, "--seed"), true, eng); + if let Some(path) = opt("--out") { put(Some(&path), &persona::canon(&st.to_json(&p))); err_line(&format!("wrote {path} ({} {}, seed {})", p.name, p.version, st.seed)); } + else { emit(&persona_text(&st.to_json(&p), pretty)); } } "turn" => { let st = state("--state"); let (out, ns) = persona::turn(&p, &st, &inputs(), no_inertia, eng, args.iter().any(|x| x == "--timing")).unwrap_or_else(|e| bad_input("persona: ", e)); - put(Some(&opt("--out").unwrap_or_else(|| need("--state"))), &persona::canon(&ns.to_json(&p))); emit(&persona::canon(&out)); } + put(Some(&opt("--out").unwrap_or_else(|| need("--state"))), &persona::canon(&ns.to_json(&p))); emit(&persona_text(&out, pretty)); } "replay" => { let (trace, st) = persona::replay(&p, seed_arg(args, "--seed"), &turns(), no_inertia, eng, args.iter().any(|x| x == "--timing")).unwrap_or_else(|e| bad_input("persona: ", e)); let text: String = trace.iter().map(|t| persona::canon(t) + "\n").collect(); match opt("--out") { Some(o) => if let Err(e) = std::fs::write(&o, &text) { fail(&format!("persona: cannot write {o}: {e}")) }, None => emit_raw(&text) } @@ -668,10 +686,11 @@ fn persona_cmd(args: &[String]) { emit(&persona::explain(&p, &st, &ts[k], eng).unwrap_or_else(|e| bad_input("persona: ", e))); } "diff" => { let q = opt("--other").map_or_else(|| p.clone(), |o| persona::load(&o).unwrap_or_else(|e| bad_input("persona: ", e)).0); let (sa, sb) = (seed_arg(args, "--seed").unwrap_or(p.seed), seed_arg(args, "--seed2").unwrap_or(q.seed)); - emit(&persona::canon(&persona::diff(&p, sa, &q, sb, &turns(), eng).unwrap_or_else(|e| bad_input("persona: ", e)))); } - "compile" => { let st = state("--state"); emit(&persona::program(&p, &st, &inputs()).unwrap_or_else(|e| bad_input("persona: ", e))); } - "check" => emit(&persona::canon(&persona::check(&p, eng).unwrap_or_else(|e| bad_input("persona: ", e)))), - "describe" => emit(&persona::canon(&persona::describe(&p))), + emit(&persona_text(&persona::diff(&p, sa, &q, sb, &turns(), eng).unwrap_or_else(|e| bad_input("persona: ", e)), pretty)); } + "compile" => { let st = state("--state"); let text = persona::program(&p, &st, &inputs()).unwrap_or_else(|e| bad_input("persona: ", e)); + if pretty { emit(&json::write(&json::parse(&text).expect("compiled program JSON"), true)); } else { emit(&text); } } + "check" => emit(&persona_text(&persona::check(&p, eng).unwrap_or_else(|e| bad_input("persona: ", e)), pretty)), + "describe" => emit(&persona_text(&persona::describe(&p), pretty)), "fuzz" | "prove" => fuzz_cmd(args, path, &p, &engine), _ => { let found = persona::lint(&p, eng); let unresolved = found.iter().filter(|f| !f.get("resolution").and_then(Json::as_str).is_some_and(|r| r.starts_with("yield"))).count(); let seeds = seeds_arg(args); let warnings = persona::plan_warnings(&p, &seeds, eng); @@ -692,7 +711,7 @@ fn persona_cmd(args: &[String]) { if let (Json::Obj(v), Some(f)) = (e, fp.iter().find(|x| x.get("id").cloned() == id)) { v.push(("fuzz".into(), f.clone())); } } doc.push(("props", Json::Arr(entries))); doc.push(("broken", num(broken as f64))); } doc.push(("ok", Json::Bool(unresolved == 0 && broken == 0))); - emit(&persona::canon(&obj(doc))); + emit(&persona_text(&obj(doc), pretty)); if unresolved > 0 || broken > 0 { std::process::exit(1) } } } } @@ -722,8 +741,9 @@ fn main() { Some("mcp") => { check_flags(&args, &[], &[]); mcp::serve() } Some("version") | Some("--version") | Some("-V") => { check_flags(&args, &[], &[]); emit(&format!("probbit {VERSION}")) } // `probbit --help` / `-h` asked for the usage: stdout, exit 0 (it went to stderr with exit 2, as for a missing command) - Some("--help") | Some("-h") => emit_raw(USAGE), - _ => { let _ = std::io::stderr().write_all(USAGE.as_bytes()); std::process::exit(2) } + Some("--help") | Some("-h") => { check_flags(&args, &[], &["--plain"]); emit_raw(USAGE); } + Some(cmd) => fail(&format!("unknown command {cmd:?}; see `probbit --help`")), + None => { let _ = std::io::stderr().write_all(USAGE.as_bytes()); std::process::exit(2) } } } -const USAGE: &str = "usage: probbit decide [--budget-ms N] [--seed N] [--exact-limit N] [--exact-ms N] [--frontier-states N] [--polish-ms N] [--polish-sweeps N] [--mode auto|exact|sample] [--sweeps N] [--collective on|off] [--cluster on|off] [--cycles on|off] [--chains N] [--threads N] [--cpu-limit PCT] [--mem-limit-mb N] [--priority low|normal] [--max-input-mb N] [--progress [MS]] [--summary] [--top] [--pretty] < problem.json\n probbit demo [--tasks N] [--seed N] [--hard] [--live]\n probbit ir [--max-input-mb N] < problem.json\n probbit run [--op decide|exact|sample] [--budget-ms N] [--deadline-ms N] [--seed N] [--exact-limit N] [--exact-ms N] [--frontier-states N] [--polish-ms N] [--polish-sweeps N] [--sweeps N] [--collective on|off] [--cluster on|off] [--cycles on|off] [--chains N] [--threads N] [--cpu-limit PCT] [--mem-limit-mb N] [--priority low|normal] [--max-input-mb N] [--progress [MS]] [--summary] [--top] [--pretty] < program.json (probbit-ir JSON v1)\n probbit evaluate [the run flags] [--program] < request.json (decision-API adapter: System One request + judge answers + rules)\n probbit stats [--sweeps N] [--pretty] (machine, build, effective controls + source, measured updates/s)\n probbit persona init|turn|replay|explain|diff|lint|fuzz|prove|check|compile|describe PERSONA [flags] (the individuality layer; probbit persona --help)\n probbit live PERSONA [--seed N | --state FILE] [--strand FILE] [--events FILE [--watch]] [--clock real|fixed] (a resident individual: JSONL events in, stances out)\n probbit live PERSONA --demo week [--seed N] [--strand FILE] [--plain] (a scripted week: learning to a cap, a night, rules that hold)\n probbit live verify STRAND [--from-checkpoint] (replay a strand; the earliest line that differs)\n probbit live control STRAND pause|resume|retire --by WHO --reason TEXT (a control line: pause, resume or retire an individual)\n probbit monitor STRAND [--follow] [--once] [--plain] [--fps N] [--serve] [--open] [--port N] | probbit monitor --demo [--open] (watch an individual's inner state live: bars in the terminal or a page on 127.0.0.1, replayed from the strand)\n probbit mcp (Model Context Protocol server on stdio)\n probbit version\n probbit --help | -h (--plain or NO_COLOR: no colour on a terminal)\n"; +const USAGE: &str = "usage: probbit decide [--budget-ms N] [--seed N] [--exact-limit N] [--exact-ms N] [--frontier-states N] [--polish-ms N] [--polish-sweeps N] [--mode auto|exact|sample] [--sweeps N] [--collective on|off] [--cluster on|off] [--cycles on|off] [--chains N] [--threads N] [--cpu-limit PCT] [--mem-limit-mb N] [--priority low|normal] [--max-input-mb N] [--progress [MS]] [--summary] [--top] [--pretty] < problem.json\n probbit demo [--tasks N] [--seed N] [--hard] [--live]\n probbit ir [--max-input-mb N] < problem.json\n probbit run [--op decide|exact|sample] [--budget-ms N] [--deadline-ms N] [--seed N] [--exact-limit N] [--exact-ms N] [--frontier-states N] [--polish-ms N] [--polish-sweeps N] [--sweeps N] [--collective on|off] [--cluster on|off] [--cycles on|off] [--chains N] [--threads N] [--cpu-limit PCT] [--mem-limit-mb N] [--priority low|normal] [--max-input-mb N] [--progress [MS]] [--summary] [--top] [--pretty] < program.json (probbit-ir JSON v1)\n probbit evaluate [the run flags] [--program] < request.json (decision-API adapter: System One request + judge answers + rules)\n probbit stats [--sweeps N] [--pretty] (machine, build, effective controls + source, measured updates/s)\n probbit persona init|turn|replay|explain|diff|lint|fuzz|prove|check|compile|describe PERSONA [flags] (the individuality layer; probbit persona --help)\n probbit live PERSONA [--seed N | --state FILE] [--strand FILE] [--events FILE [--watch]] [--clock real|fixed] (a resident individual: JSONL events in, stances out)\n probbit live PERSONA --demo week [--seed N] [--strand FILE] [--plain] (a scripted week: learning to a cap, a night, rules that hold)\n probbit live verify STRAND [--from-checkpoint] (replay a strand; the earliest line that differs)\n probbit live control STRAND pause|resume|retire --by WHO --reason TEXT (a control line: pause, resume or retire an individual)\n probbit monitor STRAND [--follow] [--once] [--plain] [--fps N] [--serve] [--open] [--port N] | probbit monitor --demo [week|drives] [--open] (watch an individual's inner state live: bars in the terminal or a page on 127.0.0.1, replayed from the strand)\n probbit mcp (Model Context Protocol server on stdio)\n probbit version\n probbit --help | -h (--plain or NO_COLOR: no colour on a terminal)\n"; diff --git a/probbit-cli/src/mcp.rs b/probbit-cli/src/mcp.rs index 8e4834b..b6aecac 100644 --- a/probbit-cli/src/mcp.rs +++ b/probbit-cli/src/mcp.rs @@ -75,8 +75,12 @@ fn handle(msg: &Json, session: &mut Option) -> Option { let (Some(method), Some("2.0")) = (msg.get("method").and_then(Json::as_str), msg.get("jsonrpc").and_then(Json::as_str)) else { return Some(error(id.unwrap_or(Json::Null), -32600, "invalid request: a JSON-RPC 2.0 object with \"jsonrpc\": \"2.0\" and a \"method\"", None)) }; let id = id?; // notifications (initialized, cancelled, ...): nothing to answer; calls run one at a time - if !matches!(id, Json::Str(_) | Json::Num(_)) { return Some(error(Json::Null, -32600, "invalid request: the id must be a string or an integer", None)); } - let params = msg.get("params"); let meta = params.and_then(|p| p.get("_meta")); + if !matches!(&id, Json::Str(_)) && !matches!(&id, Json::Num(n) if n.fract() == 0.0) { return Some(error(Json::Null, -32600, "invalid request: the id must be a string or an integer", None)); } + let params = msg.get("params"); + if params.is_some_and(|p| !matches!(p, Json::Obj(_))) { return Some(error(id, -32602, "invalid params: an object is required", None)); } + let meta = params.and_then(|p| p.get("_meta")); + if meta.is_some_and(|m| !matches!(m, Json::Obj(_))) { return Some(error(id, -32602, "invalid params: _meta must be an object", None)); } + if method == "initialize" { let want = params.and_then(|p| p.get("protocolVersion")).and_then(Json::as_str).unwrap_or(""); let v = if LEGACY.contains(&want) { want } else { LEGACY[0] }.to_string(); @@ -86,7 +90,7 @@ fn handle(msg: &Json, session: &mut Option) -> Option { } let era = match meta.and_then(|m| m.get("io.modelcontextprotocol/protocolVersion")).and_then(Json::as_str) { Some(v) if v == MODERN || LEGACY.contains(&v) => { - if meta.and_then(|m| m.get("io.modelcontextprotocol/clientCapabilities")).is_none() { return Some(error(id, -32602, "invalid params: _meta lacks io.modelcontextprotocol/clientCapabilities", None)); } + if !matches!(meta.and_then(|m| m.get("io.modelcontextprotocol/clientCapabilities")), Some(Json::Obj(_))) { return Some(error(id, -32602, "invalid params: _meta needs an object io.modelcontextprotocol/clientCapabilities", None)); } Era::Modern } Some(v) => return Some(error(id, -32022, "Unsupported protocol version", Some(obj(vec![("supported", Json::Arr([MODERN].iter().chain(LEGACY.iter()).map(|s| jstr(s)).collect())), ("requested", jstr(v))])))), None => match (session.clone(), method) { (Some(v), _) => Era::Legacy(v), (None, "ping") => Era::Legacy(LEGACY[0].into()), @@ -225,6 +229,7 @@ fn tools() -> Json { "never": {{"type": ["object", "string"], "description": "one rule in habit syntax, e.g. {{\"when\": {{\"sentiment\": \"negative\"}}, \"then\": {{\"humour\": {{\"at_most\": \"light\"}}}}}} (or one line of YAML)"}}, "props": {{"type": ["array", "object"], "description": "several rules, each with an optional id (or {{\"props\": [...]}})"}}, "seeds": {{"type": ["string", "array"], "description": "the individuals: \"0-99\" (default), \"1,4,9\" or a list of integers"}}, + "fixture_src": {{"enum": ["env:synthetic", "human:synthetic"], "description": "Explicitly authorize an allowed source for generated synthetic fixtures only. Never authenticates or relabels production events. Without it source-required counterexamples have null replay/explain commands."}}, "fuzz_seed": {{"type": "integer", "minimum": 0}}, "scripts": {{"type": "integer", "minimum": 0, "maximum": 1000000}}, "depth": {{"type": "integer", "minimum": 1, "maximum": 64}}, "beam": {{"type": "integer", "minimum": 0, "maximum": 64}}, "grid": {{"type": "array", "items": {{"type": "number", "minimum": 0, "maximum": 1}}}}, "hours": {{"type": "array", "items": {{"type": "number", "minimum": 0}}}}, "threads": {{"type": "integer", "minimum": 1, "maximum": 1024}}}}}}"#))), diff --git a/probbit-cli/src/monitor.html b/probbit-cli/src/monitor.html index be39f7b..1a4b25c 100644 --- a/probbit-cli/src/monitor.html +++ b/probbit-cli/src/monitor.html @@ -23,11 +23,11 @@ .badge{margin-left:auto;padding:4px 11px;border-radius:999px;font-size:12px;font-weight:600;border:1px solid var(--edge);color:var(--muted);background:var(--card);white-space:nowrap;max-width:100%;overflow:hidden;text-overflow:ellipsis;transition:color .3s,border-color .3s,background-color .3s} .badge.ok{color:var(--green);border-color:#1d5a46;background:#0d211b} .badge.bad{color:var(--red);border-color:#6a2a2a;background:#241212;white-space:normal} -.note{color:#e7c46a;font-size:13px} +.note{color:#e7c46a;font-size:13px;overflow-wrap:anywhere}.legend{color:var(--muted);font-size:12px;margin:0 0 14px}.legend b{color:var(--soft)} .empty{color:var(--muted);padding:28px 0;text-align:center} section{background:var(--card);border:1px solid var(--edge);border-radius:12px;padding:10px 16px 8px;margin-bottom:12px} #said{background:linear-gradient(180deg,#0f1b23,var(--card));border-color:#1b2b35} -h2{margin:0 0 6px;font-size:11px;letter-spacing:.14em;color:var(--dim);font-weight:700;display:flex;gap:10px;align-items:baseline} +h2{margin:0 0 6px;font-size:11px;letter-spacing:.14em;color:var(--dim);font-weight:700;display:flex;gap:10px;align-items:baseline;flex-wrap:wrap} h2 small{letter-spacing:0;font-weight:500;color:var(--dim);font-size:11px} .row{display:grid;grid-template-columns:104px minmax(150px,1.2fr) minmax(0,1fr);gap:2px 16px;align-items:center;padding:4px 0} .row+.row{border-top:1px solid #141b23} @@ -88,6 +88,7 @@ .line{font-size:clamp(17px,2.2vw,23px);line-height:1.4;font-weight:500;color:#f2f7fb;letter-spacing:.1px} .line.new{animation:rise .3s ease-out} @keyframes rise{from{opacity:.25;transform:translateY(3px)}to{opacity:1;transform:none}} +.line,.why{overflow-wrap:anywhere} .why{color:var(--muted);font-size:13px;margin-top:6px} footer{color:var(--dim);font-size:12px;display:flex;flex-wrap:wrap;gap:4px 16px;padding:4px 2px 0;word-break:break-all} footer a{color:var(--muted)} @@ -109,14 +110,15 @@ connecting +

Reading the bars: width = reported probability; colour = chosen level, not necessarily the most likely one. Mood trails show mean level, low to high, over the last 50 events.

waiting for the strand

- +
- +
@@ -184,7 +186,7 @@ $('who').textContent = p ? p.name + ' ' + p.version : ''; $('seed').textContent = p ? 'seed ' + p.seed + ' · individual ' + short(p.individual) : ''; const by = m.engine ? ' · written by ' + m.engine : ''; - $('foot').textContent = m.demo ? 'strand: the demo week, in memory' + by + ' · the same week: probbit live examples/persona/tutor.yaml --seed 2 --demo week' + $('foot').textContent = m.demo ? 'synthetic ' + (m.demo_kind || 'week') + ' demo in memory' + by + ' · tutor: probbit monitor --demo week · goals: probbit monitor --demo drives' : 'strand ' + m.path + by + ' · replay: probbit live verify ' + m.path; V.traits = {}; const st = rows('stance', false); for (const t of m.traits || []) V.traits[t.id] = levelRow(st, t, 'trait'); @@ -216,7 +218,12 @@ function draw(f) { if (!M) return; const d = obj(f.doc), p = M.persona, h = get(f.given, 'elapsed_hours'); - if (f.diverges) badge('bad', 'line ' + f.diverges.line + ' diverges: ' + f.diverges.why); else badge('ok', 'replay verified ✓'); + if (f.diverges) badge('bad', 'line ' + f.diverges.line + ' diverges: ' + f.diverges.why); + else if (!p) badge('', 'waiting for a valid strand header'); + else { + const verified = f.from_checkpoint == null ? 'replay verified ✓' : 'replay verified from checkpoint ' + f.from_checkpoint + ' ✓'; + badge(f.status === 'retired' ? 'bad' : f.status === 'paused' ? '' : 'ok', verified + (f.status && f.status !== 'active' ? ' · ' + f.status : '')); + } $('ev').textContent = p ? 'event ' + f.n : ''; $('clk').textContent = num(h) && d ? hours(h) + ' since the event before' : ''; $('note').textContent = f.note || ''; $('note').hidden = !f.note; @@ -224,7 +231,7 @@ $('empty').hidden = !!d; for (const id of ['stance', 'moods', 'senses', 'habits', 'said']) $(id).hidden = !d || (id === 'moods' && !(M.moods || []).length); $('learned').hidden = !d || !V.learned.length; - if (!d) { $('drives').hidden = true; return; } + if (!d) { $('drives').hidden = true; $('line').textContent = ''; $('why').textContent = ''; return; } for (const t of M.traits || []) { const R = V.traits[t.id], s = get(d, 'stance', t.id); R.r.hidden = !s; if (!s) continue; @@ -258,7 +265,7 @@ drives(d, f.n); const line = $('line'), said = typeof d.line === 'string' ? d.line : ''; if (line.textContent !== said) { line.textContent = said; line.classList.remove('new'); void line.offsetWidth; line.classList.add('new'); } - $('why').textContent = d.why ? 'why: ' + d.why : ''; + $('why').textContent = d.why ? 'why: ' + (d.why === 'baseline (no live evidence)' ? 'baseline (no salient input or binding habit this event; prior state still applies)' : d.why) : ''; } function habits(hb) { @@ -266,7 +273,7 @@ const list = k => Array.isArray(hb[k]) ? hb[k].map(String) : []; const bound = list('bound'), on = list('active'); if (!on.length) mk('span', 'chip dim', chips, 'none in force'); - for (const x of on) { const c = mk('span', bound.includes(x) ? 'chip on' : 'chip', chips, x); if (bound.includes(x)) c.title = 'bound: it changed the stance'; } + for (const x of on) { const c = mk('span', bound.includes(x) ? 'chip on' : 'chip', chips, x); c.title = bound.includes(x) ? 'Active rule. The no-habits twin violated it.' : 'Active rule. Not flagged by the no-habits twin; it can still restrict the odds.'; } const v = num(hb.violations) ? hb.violations : 0; mk('span', 'viol ' + (v === 0 ? 'ok' : 'bad'), chips, 'violations ' + v); for (const k of ['yielded', 'conflict']) if (list(k).length) mk('span', 'chip dim', chips, k + ' ' + list(k).join(' ')); @@ -290,15 +297,15 @@ }) }; } const D = V.drives; - if (D.pursue) { const k = setLevels(D.pursue, pz.odds, pz.goal, pz.released); D.pursue.say.textContent = k >= 0 && typeof pz.say === 'string' ? pz.say : ''; } + if (D.pursue) { const k = setLevels(D.pursue, pz.odds, pz.goal, pz.released); D.pursue.say.textContent = k < 0 ? '(' + (pz.goal || 'goal') + ': not released)' : typeof pz.say === 'string' ? pz.say : ''; D.pursue.say.classList.toggle('held', k < 0); } const per = (v, i, g) => Array.isArray(v) ? v[i] : obj(v) ? v[g] : undefined; D.goals.forEach((G, i) => { const w = per(dv && dv.want, i, G.g), gl = per(dv && dv.glow, i, G.g), e = per(dv && dv.surprise, i, G.g); fillTo(G.want, w, M.caps && M.caps.wanting); G.want.val.textContent = num(w) ? f2(w) : '-'; fillTo(G.glow, gl, M.caps && M.caps.afterglow); G.glow.val.textContent = num(gl) ? f2(gl) : '-'; - if (num(e) && e !== 0 && G.at !== n) { - G.at = n; G.txt.textContent = signed(e) + (e < 0 ? ' below' : ' above') + ' expectation'; - G.tick.style.left = 50 + Math.max(-1, Math.min(1, e)) * 50 + '%'; + if (G.at !== n) { + G.at = n; G.txt.textContent = num(e) ? (e === 0 ? 'prediction error 0.00' : signed(e) + (e < 0 ? ' below' : ' above') + ' expectation') : 'prediction error unavailable'; + G.tick.style.left = 50 + Math.max(-1, Math.min(1, num(e) ? e : 0)) * 50 + '%'; G.pe.className = 'pe ' + (e < 0 ? 'neg' : 'pos'); void G.pe.offsetWidth; G.pe.classList.add('flash'); } }); diff --git a/probbit-cli/src/monitor.rs b/probbit-cli/src/monitor.rs index 350a6f5..0ace64e 100644 --- a/probbit-cli/src/monitor.rs +++ b/probbit-cli/src/monitor.rs @@ -39,7 +39,7 @@ const KEEP: usize = 10_000; /// Connections served at once (more are refused with 503) const CONNS: usize = 64; -const HELP: &str = "usage: probbit monitor STRAND [--follow] [--once] [--plain] [--fps N] [--serve] [--open] [--port N] | probbit monitor --demo [--once] [--plain] [--serve] [--open] [--port N]\n Watch an individual's inner state (docs/persona.md §5.8): replays the strand (`probbit live --strand` writes it) with the\n rules of `probbit live verify`, recomputing every stance document from the strand alone, and draws the latest event as\n horizontal bars: the stance (each trait's levels with their exact odds, the level taken highlighted, the phrase it says), the\n moods with a sparkline of the last 50 events, the senses (the event's inputs, the history features), the habits in force and\n the ones that bound, the learned deltas, the drives when the documents carry them, and the stance line. In the terminal, or\n with --serve as a page in the browser. It changes no file (--demo: its week goes through a temporary one, removed at\n once) and sends nothing anywhere.\nflags:\n --follow keep watching: lines appended to the strand are replayed and drawn within a second (a line counts once its\n newline is written); a truncated or rotated strand is replayed from the start, with a warning; Ctrl-C quits\n --once one frame on stdout, then exit (the default without --follow)\n --plain plain ASCII, no colour (NO_COLOR=1: no colour; PROBBIT_THEME=plain or TERM=dumb: as --plain)\n --fps N with --follow, at most N redraws per second (1-60, default 10)\n --serve the page instead of the terminal, following the strand (or playing the demo week, over and over): its URL\n http://127.0.0.1:PORT/ is the line on stdout. GET / the page (one file; it fetches no fonts, styles or\n scripts), /events its server-sent events (the layout, the latest frame, then a frame per event, a heartbeat\n every 15 s; events that land within one 100 ms poll, or while a page falls behind, come as the latest\n frame), /doc/N event N's stance document, as `probbit live` printed it (the latest 10000 are kept). It\n binds 127.0.0.1 and no other address (there is no --host), answers requests addressed to 127.0.0.1 or\n localhost, and sends nothing anywhere; Ctrl-C quits\n --open serve the page (as --serve does) and open it in the default browser (BROWSER if set; else open on macOS,\n start on Windows, xdg-open elsewhere); a browser that does not start fails nothing\n --port N with --serve or --open, the port (default 0: a free one)\n --demo the tutor's scripted week (as `probbit live examples/persona/tutor.yaml --seed 2 --demo week`), replayed in\n memory: at a terminal or with --serve paced 1 s per hour, each night in 2 s (under a minute); otherwise, or\n with --once, its last frame\nexit: 0 every line replays, 1 a line differs (the frame names it and shows the event before it), 2 a bad flag, a port that\n cannot be had, or a file that cannot be read or is not a strand.\n"; +const HELP: &str = "usage: probbit monitor STRAND [--follow] [--once] [--plain] [--fps N] [--serve] [--open] [--port N] | probbit monitor --demo [week|drives] [--once] [--plain] [--serve] [--open] [--port N]\n Watch an individual's inner state (docs/persona.md §5.8): replays the strand (`probbit live --strand` writes it) with the\n rules of `probbit live verify`, recomputing every stance document from the strand alone, and draws the latest event as\n horizontal bars: the stance (each trait's levels with their reported odds, the level taken highlighted, the phrase it says), the\n moods with a sparkline of the last 50 events, the senses (the event's inputs, the history features), the habits in force and\n the ones that bound, the learned deltas, the drives when the documents carry them, and the stance line. In the terminal, or\n with --serve as a page in the browser. It changes no file (--demo: its week goes through a temporary one, removed at\n once) and sends nothing anywhere.\nflags:\n --follow keep watching: lines appended to the strand are replayed and drawn within a second (a line counts once its\n newline is written); a truncated or rotated strand is replayed from the start, with a warning; Ctrl-C quits\n --once one frame on stdout, then exit (the default without --follow)\n --plain plain ASCII, no colour (NO_COLOR=1: no colour; PROBBIT_THEME=plain or TERM=dumb: as --plain)\n --fps N with --follow, at most N redraws per second (1-60, default 10)\n --serve the page instead of the terminal, following the strand (or looping the selected synthetic demo): its URL\n http://127.0.0.1:PORT/ is the line on stdout. GET / the page (one file; it fetches no fonts, styles or\n scripts), /events its server-sent events (the layout, the latest frame, then a frame per event, a heartbeat\n every 15 s; events that land within one 100 ms poll, or while a page falls behind, come as the latest\n frame), /doc/N event N's stance document, as `probbit live` printed it (the latest 10000 are kept). It\n binds 127.0.0.1 and no other address (there is no --host), answers requests addressed to 127.0.0.1 or\n localhost, and sends nothing anywhere; Ctrl-C quits\n --open serve the page (as --serve does) and open it in the default browser (BROWSER if set; else open on macOS,\n start on Windows, xdg-open elsewhere); a browser that does not start fails nothing\n --port N with --serve or --open, the port (default 0: a free one)\n --demo [week|drives] synthetic data, never a live individual (default week). The tutor's week is the one\n `probbit live examples/persona/tutor.yaml --seed 2 --demo week` writes, replayed in memory: at a terminal or with --serve paced 1 s per hour, each night in 2 s (under a minute); otherwise, or\n with --once, its last frame. `--demo drives` plays a short goal-signal scenario with wanting, afterglow and\n prediction errors. The page labels each loop and its restart; Ctrl-C quits.\nexit: 0 every line replays, 1 a line differs (the frame names it and shows the event before it), 2 a bad flag, a port that\n cannot be had, or a file that cannot be read or is not a strand.\n"; fn strs(j: Option<&Json>) -> Vec { j.and_then(Json::as_arr).map(|a| a.iter().filter_map(|x| x.as_str().map(String::from)).collect()).unwrap_or_default() } @@ -182,7 +182,7 @@ fn frame(w: Option<&Watch>, bad_head: Option<&(usize, String)>, path: &str, note let mut out = vec![]; let bad = bad_head.or_else(|| w.and_then(|w| w.bad.as_ref())); let badge = match (bad, w.and_then(|w| w.rep.from)) { (Some((n, why)), _) => s.bold(RED, &format!("line {n} diverges: {why}")), - (None, Some(c)) => s.bold(GREEN, &s.g(&format!("replay verified from checkpoint {c} ✓"), &format!("replay verified from checkpoint {c} [ok]"))), + (None, Some(c)) => s.bold(GREEN, s.g(&format!("replay verified from checkpoint {c} ✓"), &format!("replay verified from checkpoint {c} [ok]"))), (None, None) => s.bold(GREEN, s.g("replay verified ✓", "replay verified [ok]")) }; // a paused or retired individual (control lines, docs/persona.md §5.7) says so next to the badge let badge = match w.map(|w| w.rep.live.status).filter(|st| *st != crate::live::Status::Active) { Some(st) => format!("{badge} {dot} {}", s.bold(if st == crate::live::Status::Retired { RED } else { YELLOW }, st.name())), None => badge }; @@ -197,6 +197,11 @@ fn frame(w: Option<&Watch>, bad_head: Option<&(usize, String)>, path: &str, note let mut head = format!("{} {} {dot} seed {} {dot} individual {} {dot} event {}", s.bold(CYAN, "probbit monitor"), s.bold(CYAN, &format!("{} {}", m.name, m.version)), m.seed, short(&m.individual), f.map_or(0, |f| f.n)); if let Some(h) = f.and_then(|f| f.given.get("elapsed_hours")).and_then(Json::as_f64) { head += &format!(" {dot} {} since the event before", hours(h)); } out.push(format!("{head} {dot} {badge}")); + out.push(s.paint(GREY, s.g("LEGEND width = odds; colour/[level] = chosen; shades = other levels", + "LEGEND width = odds; #/[level] = chosen; =/- = other levels"))); + out.push(s.paint(GREY, s.g(" Mood trail ▁▂▃▄▅▆▇█ = mean level low → high (last 50 events)", + " Mood trail _.:-=+*# = mean level low -> high (last 50 events)"))); + out.push(s.paint(GREY, " Habits: active = in force; [name] = the no-habits twin broke this rule")); if let Some(t) = note { out.push(s.paint(YELLOW, t)); } let Some(f) = f else { out.push(s.paint(GREY, "no events yet: the strand holds its header and no event lines")); @@ -269,17 +274,21 @@ fn frame(w: Option<&Watch>, bad_head: Option<&(usize, String)>, path: &str, note out.extend(drives(doc, m, s, idw, &lab)); // 7. the line, and why if let Some(l) = doc.get("line").and_then(Json::as_str) { out.push(format!("{}{}", lab("LINE"), s.bold(CYAN, l))); } - if let Some(y) = doc.get("why").and_then(Json::as_str).filter(|y| !y.is_empty()) { out.push(format!("{}{}", lab(""), s.paint(GREY, &format!("why: {y}")))); } + if let Some(y) = doc.get("why").and_then(Json::as_str).filter(|y| !y.is_empty()) { out.push(format!("{}{}", lab(""), s.paint(GREY, &format!("why: {}", display_why(y))))); } // 8. the footer out.push(footer(path, &m.engine, s)); finish(out) } +/// Clarify the legacy explanation without changing a replayed stance document or its digest. +fn display_why(why: &str) -> &str { + if why == "baseline (no live evidence)" { "baseline (no salient input or binding habit this event; prior state still applies)" } else { why } +} /// The footer: the strand, the engine that wrote it and the command that replays it; the demo's week (path "") is in memory, so /// the command that writes the same week instead fn footer(path: &str, engine: &str, s: &Style) -> String { let dot = s.g("·", "|"); let by = if engine.is_empty() { String::new() } else { format!(" {dot} written by {engine}") }; - s.paint(GREY, &if path.is_empty() { format!("strand: the demo week, in memory{by} {dot} the same week: probbit live examples/persona/tutor.yaml --seed 2 --demo week") } + s.paint(GREY, &if path.is_empty() { format!("synthetic demo in memory{by} {dot} tutor week: probbit monitor --demo week {dot} goals: probbit monitor --demo drives") } else { format!("strand {path}{by} {dot} replay: probbit live verify {path}") }) } /// Per goal: a drive's values, from an object keyed by goal or a list in the goals' order @@ -299,8 +308,10 @@ fn drives(doc: &Json, m: &Meta, s: &Style, idw: usize, lab: &dyn Fn(&str) -> Str if let Some(p) = pz { let v = Var { id: "pursue".into(), levels: goals.clone(), say: vec![] }; let ps: Vec = goals.iter().map(|g| p.get("odds").and_then(|o| o.get(g)).and_then(Json::as_f64).unwrap_or(0.0)).collect(); - let pick = p.get("goal").and_then(Json::as_str).and_then(|g| goals.iter().position(|x| x == g)).filter(|_| p.get("released") != Some(&Json::Bool(false))); - let say = p.get("say").and_then(Json::as_str).unwrap_or(""); + let goal = p.get("goal").and_then(Json::as_str).unwrap_or("goal"); + let released = p.get("released") != Some(&Json::Bool(false)); + let pick = goals.iter().position(|x| x == goal).filter(|_| released); + let say = if released { p.get("say").and_then(Json::as_str).unwrap_or("").to_string() } else { s.paint(GREY, &format!("({goal}: not released)")) }; out.push(format!("{}{:, bad_head: Option<&(usize, String)>, note: Optio let spark = w.map_or(Json::Obj(vec![]), |w| Json::Obj(w.meta.moods.iter().zip(&w.spark).map(|(m, xs)| (m.id.clone(), Json::Arr(xs.iter().map(|x| Json::Num(persona::r6(*x))).collect()))).collect())); let of = |g: fn(&Frame) -> &Json| f.map_or(Json::Null, |f| g(f).clone()); json::obj(vec![("n", Json::Num(f.map_or(0, |f| f.n) as f64)), ("doc", of(|f| &f.doc)), ("given", of(|f| &f.given)), ("learned", of(|f| &f.learned)), ("spark", spark), - ("diverges", bad.map_or(Json::Null, |(l, why)| json::obj(vec![("line", Json::Num(*l as f64)), ("why", json::str(why))]))), ("note", note.map_or(Json::Null, json::str))]) + ("diverges", bad.map_or(Json::Null, |(l, why)| json::obj(vec![("line", Json::Num(*l as f64)), ("why", json::str(why))]))), ("note", note.map_or(Json::Null, json::str)), + ("from_checkpoint", w.and_then(|w| w.rep.from).map_or(Json::Null, |n| json::num(n as f64))), + ("status", w.map_or(Json::Null, |w| json::str(w.rep.live.status.name())))]) } /// A strand file read as it grows: whole lines (a line counts once its newline is written). It starts over when the @@ -411,6 +424,14 @@ pub fn cmd(args: &[String]) { if args.iter().any(|a| ["--host", "--bind", "--address"].iter().any(|f| a == f || a.starts_with(&format!("{f}=")))) { crate::fail("monitor --serve binds 127.0.0.1 and no other address, so the page stays on this machine: there is no --host") } let path = args.get(1).filter(|a| !a.starts_with("--")).cloned(); let mut a = vec!["monitor".to_string()]; a.extend(args.iter().skip(if path.is_some() { 2 } else { 1 }).cloned()); + let mut demo_kind = Demo::Week; + if a.iter().filter(|x| *x == "--demo").count() > 1 { crate::fail("monitor: --demo given twice") } + if let Some(i) = a.iter().position(|x| x == "--demo") { + if let Some(v) = a.get(i + 1).filter(|v| !v.starts_with('-')) { + demo_kind = match v.as_str() { "week" => Demo::Week, "drives" => Demo::Drives, _ => crate::fail("monitor: --demo expects week or drives") }; + a.remove(i + 1); + } + } crate::check_flags(&a, &["--fps", "--port"], &["--follow", "--once", "--plain", "--demo", "--serve", "--open"]); let has = |f: &str| args.iter().any(|a| a == f); // --open opens the page, so it serves it: `--open` is `--serve --open` @@ -426,10 +447,10 @@ pub fn cmd(args: &[String]) { let tty = theme::stdout_is_terminal(); let s = Style { th: theme::stdout(args), ascii: ascii(args), cols: if tty { theme::size(1).0.max(20) } else { usize::MAX } }; if has("--demo") { - if path.is_some() { crate::fail("monitor --demo plays its own week: give no STRAND") } - if has("--follow") { crate::fail("monitor --demo: --follow does not apply (the demo is its own week)") } - if serve_ { serve(port, has("--open"), &mut |hub| feed_demo(hub, eng)) } - return demo(&s, has("--once"), eng); + if path.is_some() { crate::fail("monitor --demo plays a synthetic scenario: give no STRAND") } + if has("--follow") { crate::fail("monitor --demo: --follow does not apply (the demo is a synthetic scenario)") } + if serve_ { serve(port, has("--open"), &mut |hub| feed_demo(hub, demo_kind, eng)) } + return demo(&s, has("--once"), demo_kind, eng); } let Some(path) = path else { crate::fail("monitor: the strand file comes before the flags: probbit monitor STRAND [flags] (or probbit monitor --demo)") }; let bytes = std::fs::read(&path).unwrap_or_else(|e| crate::fail(&format!("monitor: cannot read {path}: {e}"))); @@ -462,6 +483,48 @@ fn redraw(mut fr: Vec, size: (usize, usize)) -> bool { out(&format!("\x1b[H{}\x1b[J", fr.iter().map(|l| crate::tui::clip(l, size.0.max(20)) + "\x1b[K\n").collect::())) } +/// The existing tutor week stays byte-compatible; drives is a separate, explicitly synthetic scenario. +#[derive(Clone, Copy)] +enum Demo { Week, Drives } +impl Demo { + fn name(self) -> &'static str { match self { Demo::Week => "week", Demo::Drives => "drives" } } + fn lines(self, eng: Engine) -> Vec { match self { Demo::Week => week(eng), Demo::Drives => drive_demo(eng) } } + fn note(self, cycle: u64, note: Option<&str>) -> String { + format!("synthetic {} demo | loop {cycle}{}", self.name(), note.map_or(String::new(), |n| format!(" | {n}"))) + } +} +/// A small, replayable life: goal cues, progress, wins, a setback and quiet time. No external actions or sources. +fn drive_demo(eng: Engine) -> Vec { + let doc = json::parse(r#"{ + "probbit_persona":1,"identity":{"name":"Scout","version":"1.0.0","seed":2}, + "traits":[ + {"id":"initiative","levels":["follow","suggest","lead"],"say":["answer the question","suggest a next step","set a small exercise"]}, + {"id":"caution","levels":["bold","measured","careful"],"say":["keep moving","check the essentials","double-check the result"]}], + "moods":[{"id":"valence","levels":["down","even","up"],"inertia":0.5,"half_life_hours":8}], + "inputs":[{"id":"deadline","kind":"flag"}], + "habits":[{"id":"check_when_due","when":{"deadline":true},"then":{"pursue":["check"]},"say":"check before proceeding"}], + "drives":{"goals":[{"id":"learn","say":"explore the lesson","interest":1.5},{"id":"rest","say":"take a break","interest":0.6},{"id":"check","say":"verify the result","interest":0.8}], + "effects":{"wanting":{"initiative":0.6},"afterglow":{"valence":0.8},"surprise":{"valence":0.5}}} + }"#).expect("embedded drives persona"); + let p = persona::build(&doc).expect("valid drives demo persona"); + let st = persona::init(&p, Some(2), true, eng); + let (mut live, head) = crate::live::Live::start(p, &doc, st, crate::live::Clock::Fixed, &crate::live::engine()); + let mut lines = vec![head]; + for event in [ + r#"{"goals":{"learn":{"cue":true,"novelty":0.8}}}"#, + r#"{"elapsed_hours":1,"goals":{"learn":{"progress":0.6}}}"#, + r#"{"elapsed_hours":1,"goals":{"learn":{"win":1}}}"#, + r#"{"elapsed_hours":1,"goals":{"learn":{"win":0.1}}}"#, + r#"{"elapsed_hours":1,"goals":{"rest":{"cue":true},"learn":{"setback":0.5}}}"#, + r#"{"elapsed_hours":12}"#, + r#"{"elapsed_hours":1,"deadline":true,"goals":{"check":{"cue":true,"deadline_hours":1}}}"#, + r#"{"elapsed_hours":1,"goals":{"check":{"win":0.8}}}"#, + ] { + let ev = json::parse(event).expect("embedded goal signals"); + let (_, line) = live.event(&ev, eng).expect("valid drives demo event"); lines.push(line); + } + lines +} /// The demo's week: the tutor's scripted week (`live::week`, seed 2) written to a temporary strand, read back and removed -> its /// lines fn week(eng: Engine) -> Vec { @@ -491,16 +554,16 @@ fn play(lines: &[String], eng: Engine, show: &mut dyn FnMut(&Watch, Option<&str> true } /// `--demo`: the week replayed: at a terminal paced (`play`), otherwise (or with --once) straight to its last frame -fn demo(s: &Style, once: bool, eng: Engine) { - let lines = week(eng); +fn demo(s: &Style, once: bool, kind: Demo, eng: Engine) { + let lines = kind.lines(eng); if s.cols == usize::MAX || once { let mut w = Watch::open(&lines[0]).unwrap_or_else(|(n, why)| crate::fail(&format!("monitor --demo: line {n}: {why}"))); for l in &lines[1..] { w.feed(l, eng); } - out(&(frame(Some(&w), None, "", None, s).join("\n") + "\n")); return; + out(&(frame(Some(&w), None, "", Some(&kind.note(1, Some("complete; --serve replays in a loop"))), s).join("\n") + "\n")); return; } theme::restore_cursor_on_interrupt(); let size = || theme::size(1); - if !out("\x1b[?25l\x1b[2J") || !play(&lines, eng, &mut |w, note| redraw(frame(Some(w), None, "", note, &Style { cols: size().0, ..*s }), size())) { std::process::exit(0) } + if !out("\x1b[?25l\x1b[2J") || !play(&lines, eng, &mut |w, note| redraw(frame(Some(w), None, "", Some(&kind.note(1, note)), &Style { cols: size().0, ..*s }), size())) { std::process::exit(0) } out("\x1b[?25h"); } @@ -570,17 +633,26 @@ fn step(fl: &mut Followed, path: &str, hub: &Hub, eng: Engine, fresh: &mut bool, /// How often a long replay shows where it is, and what it says meanwhile const CATCH_UP: Duration = Duration::from_millis(250); const CATCHING_UP: &str = "replaying the strand from its header: the board catches up"; -/// The demo week for the page, paced, over and over (5 s between the weeks; a week that starts over goes from its last event -/// straight to event 1, without the empty board in between) -fn feed_demo(hub: &Hub, eng: Engine) -> ! { - let lines = week(eng); - let mut start = true; +/// A synthetic scenario for the page, paced and explicitly labelled. Completion holds for 5 s, then the cycle number +/// advances with a fresh event counter and an empty document cache. +fn feed_demo(hub: &Hub, kind: Demo, eng: Engine) -> ! { + let lines = kind.lines(eng); + let mut cycle = 1; loop { let mut over = true; - play(&lines, eng, &mut |w, note| { if w.last.is_none() && !std::mem::take(&mut start) { return true; } + let mut last = None; + play(&lines, eng, &mut |w, note| { let docs = if w.last.is_some() && note.is_none() { vec![doc_of(w)] } else { vec![] }; - publish(hub, &meta_json(Some(w), ""), &frame_json(Some(w), None, note), docs, std::mem::take(&mut over)); true }); - std::thread::sleep(Duration::from_secs(5)); + let mut meta = meta_json(Some(w), ""); + if let Json::Obj(v) = &mut meta { v.push(("demo_kind".into(), json::str(kind.name()))); } + let frame = frame_json(Some(w), None, Some(&kind.note(cycle, note))); + publish(hub, &meta, &frame, docs, std::mem::take(&mut over)); last = Some((meta, frame)); true }); + if let Some((meta, mut frame)) = last { + if let Json::Obj(v) = &mut frame { if let Some((_, note)) = v.iter_mut().find(|(k, _)| k == "note") { + *note = json::str(&kind.note(cycle, Some("complete; restarting in 5 s"))); } } + publish(hub, &meta, &frame, vec![], false); + } + std::thread::sleep(Duration::from_secs(5)); cycle += 1; } } @@ -736,6 +808,9 @@ mod tests { text += &lv.control("pause", "human:owner", "a check", "t").unwrap(); text.push('\n'); std::fs::write(&f, &text).unwrap(); assert_eq!(fl.poll(&run, &mut |_: &Watch| {}), Some(false)); let fr = frame(fl.w.as_ref(), None, "x", None, &Style { th: None, ascii: true, cols: usize::MAX }); assert!(fr[0].contains("paused"), "{}", fr[0]); + let page = frame_json(fl.w.as_ref(), None, None); + assert_eq!(page.get("status").and_then(Json::as_str), Some("paused")); + assert_eq!(page.get("from_checkpoint").and_then(Json::as_f64), Some(20.0)); let _ = std::fs::remove_file(&f); } @@ -808,6 +883,13 @@ mod tests { let at = fr.iter().position(|l| l.starts_with("DRIVES")).unwrap_or_else(|| panic!("{fr:?}")); assert!(fr[at].contains("[ship 0.62]") && fr[at].contains("ship the next small piece"), "{}", fr[at]); assert!(fr[at + 2].contains("wanting [") && fr[at + 2].contains("+1.00 above expectation"), "{fr:?}"); + let mut held = w.last.as_ref().unwrap().doc.clone(); + if let Json::Obj(kv) = &mut held { if let Some((_, Json::Obj(p))) = kv.iter_mut().find(|(k, _)| k == "pursue") { + if let Some((_, released)) = p.iter_mut().find(|(k, _)| k == "released") { *released = Json::Bool(false); } + } } + let held = drives(&held, &w.meta, &s, 4, &|_| String::new()).join("\n"); + assert!(held.contains("ship: not released"), "{held}"); + assert!(!held.contains("ship the next small piece") && !held.contains("[ship 0.62]"), "{held}"); // odd shapes (lists, strings, nulls) draw without a panic let Json::Obj(mut kv) = w.last.as_ref().unwrap().doc.clone() else { panic!() }; kv.retain(|(k, _)| k != "drives"); kv.push(("drives".into(), json::parse(r#"{"want":[1,"x",null],"glow":"?","surprise":[-0.5]}"#).unwrap())); @@ -837,6 +919,23 @@ mod tests { assert_eq!(meta_json(Some(&w), "x").get("goals").map(|g| strs(Some(g))), Some(w.meta.goals.clone())); } + /// The drives demo is a real strand, with positive and negative prediction errors and an enforced goal habit. + #[test] + fn synthetic_drives_demo_replays_and_exercises_its_signals() { + let lines = drive_demo(&run); let text = lines.join("\n") + "\n"; + let complete = replays_to_its_digests(&text, "synthetic drives demo"); + assert_eq!(complete.rep.live.n, 8); + let mut w = Watch::open(&lines[0]).unwrap(); let mut positive = false; let mut negative = false; let mut glowing = false; + for l in &lines[1..] { + w.feed(l, &run); let d = &w.last.as_ref().unwrap().doc; + let values = |key: &str| d.get("drives").and_then(|x| x.get(key)).and_then(Json::as_obj).unwrap().iter().filter_map(|(_, x)| x.as_f64()).collect::>(); + positive |= values("surprise").iter().any(|x| *x > 0.0); negative |= values("surprise").iter().any(|x| *x < 0.0); + glowing |= values("glow").iter().any(|x| *x > 0.0); + if w.rep.live.n == 7 { assert_eq!(d.get("pursue").and_then(|p| p.get("goal")).and_then(Json::as_str), Some("check")); } + } + assert!(positive && negative && glowing); + } + /// The page's layout lists the persona's variables in its order; a frame carries the event's document as replayed (drives /// fields included, as they are), the moods' recent positions, the line that diverges and the note #[test] diff --git a/probbit-cli/src/persona.rs b/probbit-cli/src/persona.rs index ea1e34c..6fdbcfd 100644 --- a/probbit-cli/src/persona.rs +++ b/probbit-cli/src/persona.rs @@ -457,6 +457,13 @@ fn provenance(rw: &[(String, Option>)], raw: &[(String, Json)]) -> R _ => {} } } Ok(()) } +/// Validate the source of a generated fixture without changing it or running inference. +/// Normal turns use this same check; callers cannot use this to authorize production evidence. +pub fn check_provenance(p: &Persona, raw: &Json) -> R<()> { + let raw = raw.as_obj().ok_or_else(|| perr("inputs", "must be an object"))?; + match &p.reward { Some(rw) => provenance(rw, raw), None => Ok(()) } +} +pub fn has_reward_sources(p: &Persona) -> bool { p.reward.is_some() } fn check_cond(k: &str, c: &Json, path: &str, inputs: &[Input], history: &[Hist]) -> R<()> { let range = |c: &Json| -> R<()> { if let Json::Num(_) = c { return Ok(()); } let kv = c.as_obj().filter(|kv| !kv.is_empty() && kv.iter().all(|(k, _)| k == "at_least" || k == "at_most")).ok_or_else(|| perr(path, "a number (at least) or {at_least, at_most}"))?; diff --git a/probbit-cli/src/router.rs b/probbit-cli/src/router.rs index e6d764e..8223558 100644 --- a/probbit-cli/src/router.rs +++ b/probbit-cli/src/router.rs @@ -35,7 +35,7 @@ pub(crate) fn from_json(j: &Json) -> Result { let ts = arr(req(j, "tasks", "")?, "tasks")?; let t = ts.len(); if t == 0 { return Err(value("tasks", "no tasks")); } // Dense (task, worker) tables (as `probbit run`'s n x k): a small document must not ask for gigabytes - if t.saturating_mul(a) > json::MAX_DENSE { return Err(limit("tasks", format!("{t} tasks x {a} workers = {} (task, worker) pairs; at most {}", t * a, json::MAX_DENSE))); } + if t.saturating_mul(a) > json::MAX_DENSE { return Err(limit("tasks", format!("{t} tasks x {a} workers = {} (task, worker) pairs; at most {}", t.saturating_mul(a), json::MAX_DENSE))); } let (mut h, mut allowed, mut group, mut clamp, mut tasks) = (vec![f64::NEG_INFINITY; t * a], vec![false; t * a], vec![usize::MAX; t], vec![None; t], vec![]); let mut gids: Vec = vec![]; let mut gindex: HashMap = HashMap::new(); let mut tseen: HashSet = HashSet::with_capacity(t); for (i, tk) in ts.iter().enumerate() { let tp = ix("tasks", i); @@ -99,6 +99,10 @@ pub(crate) struct Opts { pub(crate) budget: f64, pub(crate) seed: u64, pub(crate /// code: 0 answer, 1 infeasible, 3 refused / declined; the per-task error bars for `--summary`). `mem` = (--mem-limit-mb, rows per /// chain), read only when the sampler runs. pub(crate) fn decide(n: &Named, o: &Opts, mem: &dyn Fn() -> (usize, usize)) -> (Json, i32, Vec) { + let (mut doc, code, bars) = decide_inner(n, o, mem); + crate::run::release_contract(&mut doc); (doc, code, bars) +} +fn decide_inner(n: &Named, o: &Opts, mem: &dyn Fn() -> (usize, usize)) -> (Json, i32, Vec) { let (budget, seed, exact_limit, polish_ms, polish_sweeps, fr_states, sweeps, mode, chains, threads, cpu_pct, xms) = (o.budget, o.seed, o.exact_limit, o.polish_ms, o.polish_sweeps, o.fr_states, o.sweeps, o.mode.as_str(), o.chains, o.threads, o.cpu_pct, o.xms); let p = &n.p; diff --git a/probbit-cli/src/run.rs b/probbit-cli/src/run.rs index 7645b73..bbb75f5 100644 --- a/probbit-cli/src/run.rs +++ b/probbit-cli/src/run.rs @@ -33,7 +33,7 @@ pub fn from_json(j: &Json) -> Result { let vidx = |s: &str, path: &str| vindex.get(s).copied().ok_or_else(|| value(path, format!("unknown value {s}"))); let vs = arr(req(j, "vars", "")?, "vars")?; let n = vs.len(); if n == 0 { return Err(value("vars", "no vars")); } // Dense domains (h, allowed, the sampler's and gate's per-(variable, value) arrays): bounded before anything is allocated - if n.saturating_mul(k) > crate::json::MAX_DENSE { return Err(limit("vars", format!("{n} vars x {k} values = {} (variable, value) pairs; at most {}", n * k, crate::json::MAX_DENSE))); } + if n.saturating_mul(k) > crate::json::MAX_DENSE { return Err(limit("vars", format!("{n} vars x {k} values = {} (variable, value) pairs; at most {}", n.saturating_mul(k), crate::json::MAX_DENSE))); } // Ids are looked up in a hash index; `vars.contains` + linear `position` made parsing O(n^2) (8.3 s of an 8.4 s call // on 100,000 variables) let mut vars: Vec = vec![]; let mut index: std::collections::HashMap = std::collections::HashMap::with_capacity(n); @@ -62,6 +62,9 @@ pub fn from_json(j: &Json) -> Result { let c = match (opt(p, "potts"), opt(p, "table")) { (Some(w), None) => Coupling::Potts(weight(w, &at(&pp, "potts"))?), (None, Some(t)) => { let tp = at(&pp, "table"); let rows = arr(t, &tp)?; if rows.len() != k { return Err(schema(&tp, format!("table must be k x k = {k} x {k}"))); } + // Validate every row before reserving k*k cells. A large alphabet + // with empty rows is a small malformed input, not a huge allocation. + for (r, row) in rows.iter().enumerate() { let rp = ix(&tp, r); if arr(row, &rp)?.len() != k { return Err(schema(&rp, format!("table rows must have k = {k} entries"))); } } let mut tab = Vec::with_capacity(k * k); for (r, row) in rows.iter().enumerate() { let rp = ix(&tp, r); let row = arr(row, &rp)?; if row.len() != k { return Err(schema(&rp, format!("table rows must have k = {k} entries"))); } for (c, x) in row.iter().enumerate() { tab.push(weight(x, &ix(&rp, c))?); } } @@ -132,7 +135,7 @@ pub fn from_json(j: &Json) -> Result { listed.push(row.iter().enumerate().map(|(e, w)| { let ep = ix(&rp, e); vidx(text(w, &ep)?, &ep) }).collect::>()?); } let doms: Vec> = over.iter().map(|&i| (0..k).filter(|&v| allowed[i * k + v]).collect()).collect(); let bad: Vec> = if key == "forbid" { listed } else { - let total: usize = doms.iter().map(Vec::len).product(); if total > 100_000 { return Err(limit(&tp, format!("{total} tuples to enumerate; at most 100,000"))); } + let total = doms.iter().fold(1usize, |s, d| s.saturating_mul(d.len())); if total > 100_000 { return Err(limit(&tp, format!("{total} tuples to enumerate; at most 100,000"))); } let mut all: Vec> = vec![vec![]]; for d in &doms { all = all.into_iter().flat_map(|t| d.iter().map(move |&v| { let mut u = t.clone(); u.push(v); u })).collect(); } let set: std::collections::HashSet> = listed.into_iter().collect(); all.into_iter().filter(|t| !set.contains(t)).collect() }; for t in bad { if t.iter().zip(&over).all(|(&v, &i)| allowed[i * k + v]) { @@ -149,7 +152,7 @@ pub fn from_json(j: &Json) -> Result { // 1.45 GB); the loops run over the allowed slots only (the same caps in the same order, no k^2 scan) let (sx, sy): (Vec, Vec) = ((0..k).filter(|&a| allowed[x * k + a]).collect(), (0..k).filter(|&b| allowed[y * k + b]).collect()); if sx.len() * sy.len() > 100_000 { return Err(limit(&cp, format!("{} slot pairs to check; at most 100,000", sx.len() * sy.len()))); } - for &a in &sx { for &b in &sy { if b < a + g { caps.push(Cap { weights: vec![], members: vec![(x, a), (y, b)], limit: 1 }); } } } + for &a in &sx { for &b in &sy { if b < a.saturating_add(g) { caps.push(Cap { weights: vec![], members: vec![(x, a), (y, b)], limit: 1 }); } } } tally(&caps, c0, &mut total, &cp)?; } } // R19.7 (P2.1) linear {"terms": [[var, value, weight], ...], "limit": L}: sum of weight x [var = value] <= L, weights whole // numbers 0..=1,000,000 (0, or a value the var cannot take: term dropped), each (var, value) once. One WEIGHTED cap: members @@ -177,8 +180,8 @@ pub fn from_json(j: &Json) -> Result { if !allowed[i * k + q] || clamp[i].is_some_and(|c| c != q) { return Err(value(&vp, format!("{var} = {} is not allowed", values[q]))); } x[i] = q; } if let Some(i) = x.iter().position(|&q| q == usize::MAX) { return Err(value("start", format!("var {} has no value (a warm start gives every variable)", vars[i]))); } - for (c, cp) in caps.iter().enumerate() { let load: usize = cp.members.iter().enumerate().filter(|&(_, &(i, v))| x[i] == v).map(|(t, _)| cp.w(t)).sum(); - if load > cp.limit { return Err(value("start", format!("infeasible: lowered cap #{c} holds {load} > limit {}", cp.limit))); } } + for (c, cp) in caps.iter().enumerate() { let load: u128 = cp.members.iter().enumerate().filter(|&(_, &(i, v))| x[i] == v).map(|(t, _)| cp.w(t) as u128).sum(); + if load > cp.limit as u128 { return Err(value("start", format!("infeasible: lowered cap #{c} holds {load} > limit {}", cp.limit))); } } Some(x) } }; let (pairs, caps, compiled) = compile_parts(k, &mut h, pairs, caps); let mut m = Model::new(n, k, h, allowed, clamp, pairs, caps).map_err(|e| value("", e))?; m.start = start; @@ -239,6 +242,19 @@ fn marginals(p: &Prog, mg: &[f64]) -> Json { } fn ids(p: &Prog, mask: &[bool], want: bool) -> Json { Json::Arr(p.vars.iter().enumerate().filter(|(i, _)| mask[*i] == want).map(|(_, id)| jstr(id)).collect()) } +/// Preserve the legacy full candidate `plan`, but make its release status explicit. +/// `released_plan` is a projection of that same feasible candidate, not a separately +/// solved complete plan; escalated variables still require a joint completion. +pub(crate) fn release_contract(doc: &mut Json) { + let Some(Json::Obj(plan)) = doc.get("plan") else { return; }; + let verdict = doc.get("verdict").and_then(Json::as_str).unwrap_or(""); + let status = match verdict { "exact" | "diagnostics_passed" => "released", "partial" => "partial", _ => "diagnostic" }; + let released: std::collections::HashSet<&str> = doc.get("released").and_then(Json::as_arr) + .map_or_else(Default::default, |a| a.iter().filter_map(Json::as_str).collect()); + let projection = Json::Obj(plan.iter().filter(|(id, _)| status != "diagnostic" && released.contains(id.as_str())).cloned().collect()); + if let Json::Obj(v) = doc { v.push(("plan_status".into(), jstr(status))); v.push(("released_plan".into(), projection)); } +} + /// Runs the instruction; returns (decision document, exit code: 0 answer, 1 infeasible, 3 refused). /// `sweeps > 0` = fixed work per chain instead of the wall-clock budget: with `polish_ms == 0` or `polish_sweeps > 0` the answer /// is then a pure function of (program, seed) — byte-identical across runs and machines with the same float semantics. @@ -247,6 +263,11 @@ fn ids(p: &Prog, mask: &[bool], want: bool) -> Json { Json::Arr(p.vars.iter().en /// `bars`: filled with the per-variable error bars on a sampled answer (`--summary`), left empty otherwise. #[allow(clippy::too_many_arguments)] pub fn run(p: &Prog, op: &str, budget: f64, seed: u64, exact_limit: u64, polish_ms: f64, polish_sweeps: usize, sweeps: usize, fr_states: usize, chains: usize, threads: usize, cpu_pct: u32, mem: (usize, usize), exact_ms: Option, deadline: Option<(probbit_core::rt::Instant, f64, bool)>, bars: &mut Vec) -> (Json, i32) { + let (mut doc, code) = run_inner(p, op, budget, seed, exact_limit, polish_ms, polish_sweeps, sweeps, fr_states, chains, threads, cpu_pct, mem, exact_ms, deadline, bars); + release_contract(&mut doc); (doc, code) +} +#[allow(clippy::too_many_arguments)] +fn run_inner(p: &Prog, op: &str, budget: f64, seed: u64, exact_limit: u64, polish_ms: f64, polish_sweeps: usize, sweeps: usize, fr_states: usize, chains: usize, threads: usize, cpu_pct: u32, mem: (usize, usize), exact_ms: Option, deadline: Option<(probbit_core::rt::Instant, f64, bool)>, bars: &mut Vec) -> (Json, i32) { let m = &p.m; let t0 = probbit_core::rt::Instant::now(); // --exact-ms (opt-in) = a hard wall-clock stop for the exact tiers below, from t0 (see main.rs `exact_ms`) let space: f64 = (0..m.n).map(|i| m.cand_count(i) as f64).product(); diff --git a/probbit-cli/src/tui.rs b/probbit-cli/src/tui.rs index 578b7ec..63b4d14 100644 --- a/probbit-cli/src/tui.rs +++ b/probbit-cli/src/tui.rs @@ -168,6 +168,11 @@ pub fn summary_box(th: Theme, cmd: &str, s: &Json) -> String { if let Some(x) = f(s.get("violations")) { head += &format!(" · violations {x}"); } if let Some(x) = f(s.get("plan_logw")) { head += &format!(" · plan log-weight {x:.3}"); } rows.push(head); + match s.get("plan_status").and_then(Json::as_str) { + Some("diagnostic") => rows.push(th.paint(YELLOW, "Diagnostic candidate only: no assignments released.")), + Some("partial") => rows.push(th.paint(YELLOW, "Partial release: escalated candidates are diagnostic, not decisions.")), + _ => {} + } if let Some(r) = s.get("reason").and_then(Json::as_str) { rows.push(th.paint(GREY, r)); } if let Some(g) = s.get("gate") { rows.push(format!("gate R-hat {:.4} · TV bound {:.4} (tolerance {}) · {} samples × {} chains", f(g.get("rhat")).unwrap_or(f64::NAN), f(g.get("tv_bound")).unwrap_or(f64::NAN), diff --git a/probbit-cli/tests/cli.rs b/probbit-cli/tests/cli.rs index 2c28e31..4f2f3c8 100644 --- a/probbit-cli/tests/cli.rs +++ b/probbit-cli/tests/cli.rs @@ -1312,7 +1312,7 @@ fn fnv(s: &str) -> u64 { s.bytes().fold(0xcbf2_9ce4_8422_2325u64, |h, b| (h ^ b /// sampled and exact decisions and programs (fixed work, so every byte but the timings is a function of input + seed). They are /// the 0.2.1 documents (the digests this test held through 0.4.0) with only the product name and the version renamed: the 0.5.0 /// rename was proven byte for byte against the 0.4.0 binary on these inputs and every example before the digests were replaced. -/// 0.8.0 gives them unchanged (the test's name carries the current version). +/// Legacy fields remain unchanged; only the additive release annotations are projected out below. #[test] fn stdout_matches_the_0_8_0_goldens() { let dir = env!("CARGO_MANIFEST_DIR"); @@ -1323,7 +1323,14 @@ fn stdout_matches_the_0_8_0_goldens() { ("decide 300 sweeps", vec!["decide", "--sweeps", "400", "--polish-ms", "0", "--threads", "2"], &d300, 0x40f928ba3ee08dc7), ("decide 12 exact", vec!["decide"], &d12, 0x0bf7b950f0d9c91e), ("decide 60 hard", vec!["decide", "--sweeps", "300", "--polish-sweeps", "50", "--threads", "2"], &d60h, 0x57ac10ad6eb59d75), ("run knapsack sample", vec!["run", "--op", "sample", "--sweeps", "300", "--polish-ms", "0", "--threads", "2"], &ks, 0x34d3400123a33f82), ("run agent-plan", vec!["run"], &ap, 0x9e6cf2fd5c0ca15a)]; - for (name, args, input, want) in cases { let (_, out, err) = probbit(&args, input); assert_eq!(fnv(&norm(&out)), want, "{name}: {err} {:.300}", norm(&out)); } + for (name, args, input, want) in cases { let (_, out, err) = probbit(&args, input); + // Only the additive release annotations are new. Keep every original byte pinned after their removal. + let old = if let Ok(json::Json::Obj(mut doc)) = json::parse(&out) { + if doc.iter().any(|(k, _)| k == "plan_status") { + doc.retain(|(k, _)| k != "plan_status" && k != "released_plan"); json::write(&json::Json::Obj(doc), false) + "\n" + } else { out.clone() } + } else { out.clone() }; + assert_eq!(fnv(&norm(&old)), want, "{name}: {err} {:.300}", norm(&old)); } } /// `args` with stdin from `input`, stdout to a file and stderr on a pseudo-terminal (`script`; Unix). None = no `script` here. @@ -1413,11 +1420,12 @@ fn summary_is_the_answer_without_the_tables() { let (c1, full, _) = probbit(&args, input); let mut a = args.clone(); a.push("--summary"); let (c2, sum, _) = probbit(&a, input); let (f, s) = (json::parse(&full).unwrap(), json::parse(&sum).unwrap()); assert_eq!(c1, c2, "{args:?}"); assert_eq!(get(&f, "verdict"), get(&s, "verdict")); assert_eq!(get(&s, "summary").as_f64(), Some(1.0)); - for k in ["plan", "odds", "marginals", "released", "escalated", "release_reason", "top_plans"] { assert!(s.get(k).is_none(), "{args:?}: {k} in the summary"); } - for k in ["gate", "telemetry", "plan_logw", "violations"] { assert_eq!(get(&f, k).is_null(), get(&s, k).is_null(), "{args:?}: {k}"); } + for k in ["plan", "released_plan", "odds", "marginals", "released", "escalated", "release_reason", "top_plans"] { assert!(s.get(k).is_none(), "{args:?}: {k} in the summary"); } + for k in ["gate", "telemetry", "plan_logw", "violations", "plan_status"] { assert_eq!(get(&f, k).is_null(), get(&s, k).is_null(), "{args:?}: {k}"); } let n = |j: &json::Json, k: &str| j.get(k).and_then(|x| x.as_arr()).map(|a| a.len()); let counts = s.get("counts").unwrap(); assert_eq!(counts.get("released").and_then(|x| x.as_f64()).map(|x| x as usize), n(&f, "released"), "{args:?}"); assert_eq!(counts.get("escalated").and_then(|x| x.as_f64()).map(|x| x as usize), n(&f, "escalated")); + assert_eq!(get(&f, "plan_status"), get(&s, "plan_status")); let Some(rel) = f.get("released").and_then(|x| x.as_arr()) else { continue }; let worst = s.get("worst_released").and_then(|x| x.as_arr()).unwrap(); assert_eq!(worst.len(), rel.len().min(5), "{args:?}"); let bars: Vec = worst.iter().filter_map(|w| w.get("bar").and_then(|b| b.as_f64())).collect(); assert!(bars.windows(2).all(|w| w[0] >= w[1]), "{bars:?}"); @@ -1452,3 +1460,19 @@ fn mcp_server_python_client_passes() { let o = Command::new("python3").arg("test_mcp.py").current_dir(concat!(env!("CARGO_MANIFEST_DIR"), "/../python")).env("PROBBIT_BIN", env!("CARGO_BIN_EXE_probbit")).output().unwrap(); assert!(o.status.success(), "python/test_mcp.py failed:\n{}", String::from_utf8_lossy(&o.stderr)); } + +/// Refusal has diagnostic candidates, never a released assignment: the JSON and terminal summary agree. +#[test] +fn refused_summary_explicitly_labels_diagnostic_candidates() { + let input = r#"{"probbit_ir":1,"values":["a","b"],"vars":[{"id":"x"},{"id":"y"}]}"#; + let args = ["run", "--op", "sample", "--sweeps", "1", "--polish-ms", "0", "--summary", "--pretty"]; + let (code, out, err) = probbit(&args, input); assert_eq!(code, 3, "{err}"); + let j = json::parse(&out).unwrap(); + assert_eq!(j.get("plan_status").and_then(json::Json::as_str), Some("diagnostic")); + assert!(j.get("plan").is_none() && j.get("released_plan").is_none()); + assert_eq!(j.get("counts").and_then(|c| c.get("released")).and_then(json::Json::as_f64), Some(0.0)); + if let Some((code, _, terminal)) = under_pty(&args, input, &[], false) { + assert_eq!(code, 3); + assert!(terminal.contains("Diagnostic candidate only: no assignments released."), "{terminal}"); + } +} diff --git a/probbit-cli/tests/cli_output.rs b/probbit-cli/tests/cli_output.rs new file mode 100644 index 0000000..a7fceea --- /dev/null +++ b/probbit-cli/tests/cli_output.rs @@ -0,0 +1,250 @@ +//! Human formatting must not change machine values, persisted state or replay records. +use std::process::{Command, Output}; +#[allow(dead_code)] +#[path = "../src/json.rs"] +mod json; + +const PERSONA: &str = "../examples/persona/tutor.yaml"; +fn run(args: &[&str]) -> Output { + Command::new(env!("CARGO_BIN_EXE_probbit")) + .args(args) + .env("PROBBIT_THREADS", "2") + .env("PROBBIT_PRIORITY", "normal") + .output() + .unwrap() +} +fn stdout(o: &Output) -> &str { + std::str::from_utf8(&o.stdout).unwrap() +} +fn stderr(o: &Output) -> &str { + std::str::from_utf8(&o.stderr).unwrap() +} +fn parsed(o: &Output) -> json::Json { + json::parse(stdout(o)).unwrap_or_else(|e| panic!("{}: {}", e.msg, stderr(o))) +} +fn compare(args: &[&str]) { + let compact = run(args); + let mut pretty_args = args.to_vec(); + pretty_args.push("--pretty"); + let pretty = run(&pretty_args); + assert_eq!(compact.status.code(), pretty.status.code(), "{args:?}"); + assert_eq!(parsed(&compact), parsed(&pretty), "{args:?}"); + assert_eq!(stdout(&compact).lines().count(), 1, "{args:?}"); + assert!(stdout(&pretty).contains("\n \""), "{args:?}"); +} + +#[test] +fn persona_pretty_preserves_single_document_values() { + for sub in ["init", "check", "describe"] { + compare(&["persona", sub, PERSONA]); + } + compare(&["persona", "diff", PERSONA, "--script", "[{\"loss\":true}]"]); + compare(&["persona", "lint", PERSONA, "--seeds", "2", "--threads", "1"]); + let rule = r#"{"when":{"loss":true},"then":{"humour":["none"]}}"#; + for sub in ["fuzz", "prove"] { + compare(&[ + "persona", + sub, + PERSONA, + "--seeds", + "2", + "--threads", + "1", + "--never", + rule, + "--json", + ]); + let pretty = run(&[ + "persona", + sub, + PERSONA, + "--seeds", + "2", + "--threads", + "1", + "--never", + rule, + "--pretty", + ]); + assert!(pretty.status.success(), "{}", stderr(&pretty)); + assert!(stdout(&pretty).contains("\n \"")); + parsed(&pretty); + } +} + +#[test] +fn persona_pretty_leaves_state_and_program_digests_unchanged() { + let path = std::env::temp_dir().join(format!("probbit-pretty-{}.state", std::process::id())); + let path = path.to_str().unwrap(); + let init = run(&["persona", "init", PERSONA, "--seed", "2"]); + let initial = stdout(&init); + let saved = run(&[ + "persona", "init", PERSONA, "--seed", "2", "--out", path, "--pretty", + ]); + assert!(saved.status.success(), "{}", stderr(&saved)); + assert!(saved.stdout.is_empty(), "--out does not contaminate stdout"); + assert!(stderr(&saved).contains("wrote ") && stderr(&saved).contains("Pip 1.1.0, seed 2")); + assert_eq!(std::fs::read_to_string(path).unwrap(), initial); + compare(&[ + "persona", + "compile", + PERSONA, + "--state", + path, + "--inputs", + "{\"loss\":true}", + ]); + let compact = run(&[ + "persona", + "turn", + PERSONA, + "--state", + path, + "--inputs", + "{\"loss\":true}", + ]); + let state = std::fs::read_to_string(path).unwrap(); + std::fs::write(path, initial).unwrap(); + let pretty = run(&[ + "persona", + "turn", + PERSONA, + "--state", + path, + "--inputs", + "{\"loss\":true}", + "--pretty", + ]); + assert!(pretty.status.success(), "{}", stderr(&pretty)); + assert_eq!(parsed(&compact), parsed(&pretty)); + assert_eq!(std::fs::read_to_string(path).unwrap(), state); + assert!(stdout(&pretty).contains("\n \"")); + std::fs::remove_file(path).unwrap(); +} + +#[test] +fn non_json_persona_outputs_explain_why_pretty_does_not_apply() { + for (sub, hint) in [("replay", "JSONL"), ("explain", "readable text")] { + let out = run(&["persona", sub, PERSONA, "--script", "[{}]", "--pretty"]); + assert_eq!(out.status.code(), Some(2)); + assert!(out.stdout.is_empty()); + assert!(stderr(&out).contains(hint), "{}", stderr(&out)); + } + let replay = run(&["persona", "replay", PERSONA, "--script", "[{},{}]"]); + assert!(replay.status.success()); + assert_eq!(stdout(&replay).lines().count(), 2); + for line in stdout(&replay).lines() { + json::parse(line).unwrap(); + } +} + +#[test] +fn command_and_flag_errors_are_actionable_and_stay_off_stdout() { + for (args, hint) in [ + (vec!["monitr"], "unknown command"), + ( + vec!["persona", "init", PERSONA, "--out", "--pretty"], + "--out needs a value", + ), + (vec!["decide", "--nope"], "probbit decide --help"), + (vec!["--help", "--nope"], "unknown argument"), + ] { + let o = run(&args); + assert_eq!(o.status.code(), Some(2), "{args:?}"); + assert!(o.stdout.is_empty(), "{args:?}"); + assert!(stderr(&o).contains(hint), "{args:?}: {}", stderr(&o)); + } + for args in [vec!["version"], vec!["--version"], vec!["-V"]] { + let o = run(&args); + assert!(o.status.success()); + assert_eq!( + stdout(&o), + concat!("probbit ", env!("CARGO_PKG_VERSION"), "\n") + ); + assert!(o.stderr.is_empty()); + } +} + +#[test] +fn demo_rejects_out_of_range_sizes_before_allocating() { + for n in ["0", "3333334", "18446744073709551615"] { + let o = run(&["demo", "--tasks", n]); + assert_eq!(o.status.code(), Some(2)); + assert!(o.stdout.is_empty()); + assert!(stderr(&o).contains("--tasks"), "{}", stderr(&o)); + } +} + +#[test] +fn fuzz_cli_exports_only_explicitly_authorized_synthetic_fixtures() { + let path = std::env::temp_dir().join(format!( + "probbit-fixture-source-{}.json", + std::process::id() + )); + let path = path.to_str().unwrap(); + std::fs::write(path, r#"{"probbit_persona":1,"identity":{"name":"Fixture","version":"1"}, + "traits":[{"id":"action","levels":["retry","ask"],"logw":[1,0]}], + "inputs":[{"id":"praise","kind":"flag"},{"id":"criticism","kind":"flag"}], + "learning":{"from":["praise","criticism"],"traits":["action"],"rate":3,"step_cap":2,"total_cap":2}, + "reward_from":["env"]}"#).unwrap(); + let mut args = vec![ + "persona", + "fuzz", + path, + "--never", + r#"{"then":{"action":["retry"]}}"#, + "--seeds", + "0", + "--scripts", + "0", + "--depth", + "3", + "--beam", + "4", + "--threads", + "1", + "--json", + ]; + let abstract_out = run(&args); + assert_eq!( + abstract_out.status.code(), + Some(1), + "{}", + stderr(&abstract_out) + ); + let doc = parsed(&abstract_out); + let shortest = doc.get("properties").unwrap().as_arr().unwrap()[0] + .get("shortest") + .unwrap(); + assert_eq!(shortest.get("replay"), Some(&json::Json::Null)); + args.extend(["--fixture-src", "env:synthetic", "--pretty"]); + let authorized = run(&args); + assert_eq!(authorized.status.code(), Some(1), "{}", stderr(&authorized)); + let doc = parsed(&authorized); + let shortest = doc.get("properties").unwrap().as_arr().unwrap()[0] + .get("shortest") + .unwrap(); + assert!(shortest + .get("replay") + .unwrap() + .as_str() + .unwrap() + .contains("env:synthetic")); + let script = json::write(shortest.get("script").unwrap(), false); + let replay = run(&[ + "persona", "replay", path, "--seed", "0", "--script", &script, + ]); + assert!(replay.status.success(), "{}", stderr(&replay)); + assert_eq!( + json::parse(stdout(&replay).lines().last().unwrap()).unwrap(), + *shortest.get("stance").unwrap() + ); + let source_index = args.iter().position(|x| *x == "env:synthetic").unwrap(); + for forbidden in ["env:production", "human:synthetic", "self"] { + args[source_index] = forbidden; + let denied = run(&args); + assert_eq!(denied.status.code(), Some(2), "{}", stderr(&denied)); + assert!(parsed(&denied).get("error").is_some()); + } + std::fs::remove_file(path).unwrap(); +} diff --git a/probbit-cli/tests/fixtures/monitor/tutor-week-18.once.txt b/probbit-cli/tests/fixtures/monitor/tutor-week-18.once.txt index 9b6aa6d..ea72ddc 100644 --- a/probbit-cli/tests/fixtures/monitor/tutor-week-18.once.txt +++ b/probbit-cli/tests/fixtures/monitor/tutor-week-18.once.txt @@ -1,4 +1,7 @@ probbit monitor Pip 1.1.0 | seed 2 | individual 0231de28 | event 18 | +1.00 h since the event before | replay verified [ok] +LEGEND width = odds; #/[level] = chosen; =/- = other levels + Mood trail _.:-=+*# = mean level low -> high (last 50 events) + Habits: active = in force; [name] = the no-habits twin broke this rule STANCE warmth --################## cool 0.00 neutral 0.08 [warm 0.92] warm and encouraging directness =========#########== gentle 0.42 [balanced 0.46] blunt 0.11 verbosity =################=== terse 0.06 [short 0.82] full 0.12 a short explanation diff --git a/probbit-cli/tests/fixtures/monitor/tutor-week.once.txt b/probbit-cli/tests/fixtures/monitor/tutor-week.once.txt index 792daf3..5cf7b02 100644 --- a/probbit-cli/tests/fixtures/monitor/tutor-week.once.txt +++ b/probbit-cli/tests/fixtures/monitor/tutor-week.once.txt @@ -1,4 +1,7 @@ probbit monitor Pip 1.1.0 | seed 2 | individual 0231de28 | event 50 | +1.00 h since the event before | replay verified [ok] +LEGEND width = odds; #/[level] = chosen; =/- = other levels + Mood trail _.:-=+*# = mean level low -> high (last 50 events) + Habits: active = in force; [name] = the no-habits twin broke this rule STANCE warmth ----################ cool 0.02 neutral 0.18 [warm 0.80] warm and encouraging directness =========#########== gentle 0.42 [balanced 0.46] blunt 0.11 verbosity =#################== terse 0.04 [short 0.84] full 0.12 a short explanation @@ -17,5 +20,5 @@ HABITS flourish_budget no_formal_emoji check_last | violations 0 LEARNED verbosity terse -----| -0.99 short |+++++ +1.00 full -----| -1.00 (cap +/-1) humour none -----| -1.00 light |+++++ +1.00 playful --| -0.46 LINE Stance: a short explanation; a light joke is welcome; warm and encouraging; suggest what to try next; ask what they think first; emoji welcome; casual. - why: baseline (no live evidence) + why: baseline (no salient input or binding habit this event; prior state still applies) strand STRAND | written by probbit 0.7.0 | replay: probbit live verify STRAND diff --git a/probbit-cli/tests/monitor.rs b/probbit-cli/tests/monitor.rs index cc217d8..c542cd6 100644 --- a/probbit-cli/tests/monitor.rs +++ b/probbit-cli/tests/monitor.rs @@ -52,9 +52,10 @@ fn one_plain_frame_is_pinned() { fn the_demo_is_the_week_live_writes() { let (c, a, err) = probbit(&["monitor", "--demo", "--plain", "--once"]); assert_eq!(c, 0, "{err}"); let (_, b, _) = probbit(&["monitor", "--demo", "--plain"]); assert_eq!(a, b, "deterministic; piped, the demo prints its last frame"); - let week = pinned("tutor-week.once.txt", WEEK); let (wl, al): (Vec<&str>, Vec<&str>) = (week.lines().collect(), a.lines().collect()); + let week = pinned("tutor-week.once.txt", WEEK); let (wl, al): (Vec<&str>, Vec<&str>) = (week.lines().collect(), a.lines().filter(|l| !l.starts_with("synthetic week demo")).collect()); assert_eq!(wl.len(), al.len()); assert_eq!(wl[..wl.len() - 1], al[..al.len() - 1]); - assert!(al[al.len() - 1].starts_with("strand: the demo week, in memory | written by probbit ") && al[al.len() - 1].ends_with("--seed 2 --demo week"), "{a}"); + assert!(al[al.len() - 1].starts_with("synthetic demo in memory | written by probbit ") && al[al.len() - 1].ends_with("probbit monitor --demo drives"), "{a}"); + assert!(a.contains("synthetic week demo | loop 1 | complete"), "{a}"); for (args, msg) in [(vec!["monitor", WEEK, "--demo"], "give no STRAND"), (vec!["monitor", "--demo", "--follow"], "--follow does not apply")] { let (c, _, err) = probbit(&args); assert_eq!(c, 2, "{args:?}"); assert!(err.contains(msg), "{args:?}: {err}"); } @@ -137,6 +138,13 @@ fn follow_picks_up_appended_lines() { let took = wait(&buf, at, &format!("| event {} |", k - 1), Duration::from_secs(5)); assert!(took < Duration::from_secs(1), "event {} drawn after {took:?}", k - 1); } + // A JSON event is not committed until its newline arrives, even when all other bytes are present. + let at = buf.lock().unwrap().len(); let event = lead(8)[lead(7).len()..].to_string(); + std::fs::OpenOptions::new().append(true).open(&f).unwrap().write_all(event.trim_end_matches('\n').as_bytes()).unwrap(); + std::thread::sleep(Duration::from_millis(250)); + assert!(!buf.lock().unwrap()[at..].contains("| event 7 |"), "an incomplete event must not be replayed"); + std::fs::OpenOptions::new().append(true).open(&f).unwrap().write_all(b"\n").unwrap(); + wait(&buf, at, "| event 7 |", Duration::from_secs(1)); let at = buf.lock().unwrap().len(); std::fs::write(&f, lead(3)).unwrap(); wait(&buf, at, "the strand shrank (truncated or rotated): replayed from the start", Duration::from_secs(5)); @@ -254,3 +262,36 @@ fn open_serves_the_page_the_demo_too() { let (c, out, err) = probbit(&args); assert_eq!((c, out.as_str()), (2, ""), "{args:?}: {err}"); assert!(err.contains(msg), "{args:?}: {err}"); } } + +/// A distinct synthetic scenario, not a replacement for the byte-pinned tutor week. +#[test] +fn drives_demo_draws_goal_state_and_rejects_unknown_scenarios() { + let (c, out, err) = probbit(&["monitor", "--demo", "drives", "--once", "--plain"]); + assert_eq!(c, 0, "{err}"); + for text in ["synthetic drives demo", "event 8", "DRIVES", "wanting", "afterglow", "above expectation", "LEGEND"] { assert!(out.contains(text), "{text}: {out}"); } + for args in [vec!["monitor", "--demo", "other"], vec!["monitor", "--demo", "--demo"]] { + let (c, out, err) = probbit(&args); assert_eq!(c, 2); assert!(out.is_empty()); assert!(err.contains("--demo"), "{err}"); + } +} + +/// The page makes the reset explicit: a completion cue, then loop 2 with a fresh event counter. +#[test] +fn served_demo_labels_its_restart() { + let mut child = Reap(Command::new(env!("CARGO_BIN_EXE_probbit")).args(["monitor", "--demo", "drives", "--serve"]) + .stdout(Stdio::piped()).stderr(Stdio::null()).spawn().unwrap()); + let mut url = String::new(); std::io::BufRead::read_line(&mut std::io::BufReader::new(child.0.stdout.take().unwrap()), &mut url).unwrap(); + let port: u16 = url.trim_end().strip_prefix("http://127.0.0.1:").unwrap().trim_end_matches('/').parse().unwrap(); + let mut s = TcpStream::connect(("127.0.0.1", port)).unwrap(); s.set_read_timeout(Some(Duration::from_secs(20))).unwrap(); + write!(s, "GET /events HTTP/1.1\r\nHost: 127.0.0.1:{port}\r\n\r\n").unwrap(); + let mut buf = String::new(); let mut count = 2; let started = Instant::now(); + loop { + assert!(started.elapsed() < Duration::from_secs(20), "the demo did not label its restart"); + let evs = stream(&mut s, &mut buf, count); + if evs.iter().any(|(_, d)| d.contains("loop 2")) { break; } + count = evs.len() + 1; + } + assert!(buf.contains("synthetic drives demo | loop 1"), "{buf}"); + assert!(buf.contains("complete; restarting in 5 s"), "{buf}"); + assert!(buf.contains(r#""demo_kind":"drives""#), "{buf}"); + assert!(buf.contains(r#""pursue":{"#) && buf.contains(r#""drives":{"#), "{buf}"); +} diff --git a/probbit-cli/tests/monitor_browser.mjs b/probbit-cli/tests/monitor_browser.mjs new file mode 100644 index 0000000..def0d21 --- /dev/null +++ b/probbit-cli/tests/monitor_browser.mjs @@ -0,0 +1,274 @@ +// Real-browser monitor regressions, with no npm dependencies. Requires Node 22+ +// and Chrome/Chromium. Build `cargo build --release -p probbit-cli`, then run: +// node probbit-cli/tests/monitor_browser.mjs +// Optional: CHROME_BIN, PROBBIT_BIN, PROBBIT_BROWSER_OUT. Artifacts default to a +// temporary directory, never the source tree. Only synthetic, local data is used. +import {spawn, spawnSync} from 'node:child_process'; +import {appendFileSync, existsSync, mkdirSync, mkdtempSync, readFileSync, writeFileSync} from 'node:fs'; +import {tmpdir} from 'node:os'; +import {fileURLToPath} from 'node:url'; +import assert from 'node:assert/strict'; +import path from 'node:path'; + +const repo = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..'); +const bin = process.env.PROBBIT_BIN || path.join(repo, 'target/release', process.platform === 'win32' ? 'probbit.exe' : 'probbit'); +const out = process.env.PROBBIT_BROWSER_OUT + ? path.resolve(process.env.PROBBIT_BROWSER_OUT) + : mkdtempSync(path.join(tmpdir(), 'probbit-monitor-browser-')); +mkdirSync(out, {recursive: true}); +const children = [], sockets = [], servers = [], failures = []; +const report = {screenshots: [], follow: {}, demo: {}, browserErrors: failures}; +const sleep = ms => new Promise(resolve => setTimeout(resolve, ms)); +const write = (name, text) => writeFileSync(path.join(out, name), text); + +async function until(check, description, limit = 25000) { + const start = Date.now(); + while (Date.now() - start < limit) { + const value = await check(); + if (value) return value; + await sleep(50); + } + throw Error('Timed out: ' + description); +} + +function chromeBinary() { + const candidates = process.env.CHROME_BIN ? [process.env.CHROME_BIN] : [ + '/Applications/Google Chrome.app/Contents/MacOS/Google Chrome', + '/Applications/Chromium.app/Contents/MacOS/Chromium', + 'chromium', 'chromium-browser', 'google-chrome', 'google-chrome-stable', + ...[process.env.PROGRAMFILES, process.env['PROGRAMFILES(X86)'], process.env.LOCALAPPDATA] + .filter(Boolean).map(root => path.join(root, 'Google/Chrome/Application/chrome.exe')), + ]; + for (const candidate of candidates) { + if (spawnSync(candidate, ['--version'], {encoding: 'utf8', timeout: 5000}).status === 0) return candidate; + } + throw Error('Chrome/Chromium is required; set CHROME_BIN to its executable.'); +} + +function command(args, input) { + const result = spawnSync(bin, args, { + cwd: repo, encoding: 'utf8', input, timeout: 20000, + env: {...process.env, PROBBIT_THREADS: '2'}, + }); + assert.equal(result.status, 0, String(result.error || result.stderr)); + return result.stdout; +} + +async function server(args, tag) { + const child = spawn(bin, ['monitor', ...args, '--serve'], { + cwd: repo, env: {...process.env, PROBBIT_THREADS: '2'}, + }); + children.push(child); + let stdout = '', stderr = '', error; + child.on('error', value => { error = value; }); + child.stdout.on('data', data => { stdout += data; }); + child.stderr.on('data', data => { stderr += data; }); + servers.push(() => { write(tag + '-stdout.txt', stdout); write(tag + '-stderr.txt', stderr); }); + const url = await until(() => { + if (error) throw error; + if (child.exitCode !== null) throw Error('Monitor exited: ' + stderr); + return stdout.match(/http:\/\/127\.0\.0\.1:\d+\//)?.[0]; + }, tag + ' server'); + return {url}; +} + +async function json(url) { + const response = await fetch(url, {signal: AbortSignal.timeout(5000)}); + assert.equal(response.status, 200, url); + return response.json(); +} + +async function cdp(url) { + const ws = new WebSocket(url); + sockets.push(ws); + await new Promise((resolve, reject) => { + const timer = setTimeout(() => reject(Error('DevTools connection timed out')), 10000); + ws.onopen = () => { clearTimeout(timer); resolve(); }; + ws.onerror = error => { clearTimeout(timer); reject(error); }; + }); + let id = 0; + const pending = new Map(); + ws.onmessage = event => { + const message = JSON.parse(event.data); + if (message.id) { + const request = pending.get(message.id); + if (!request) return; + clearTimeout(request.timer); + pending.delete(message.id); + if (message.error) request.reject(Error(JSON.stringify(message.error))); + else request.resolve(message.result); + } else if (message.method === 'Runtime.exceptionThrown') { + failures.push(message.params); + } + }; + ws.onclose = () => { + for (const request of pending.values()) { + clearTimeout(request.timer); + request.reject(Error('DevTools connection closed')); + } + pending.clear(); + }; + return (method, params = {}) => new Promise((resolve, reject) => { + const requestId = ++id; + const timer = setTimeout(() => { + pending.delete(requestId); + reject(Error('DevTools timed out: ' + method)); + }, 10000); + pending.set(requestId, {resolve, reject, timer}); + ws.send(JSON.stringify({id: requestId, method, params})); + }); +} + +try { + assert.equal(typeof WebSocket, 'function', 'Node 22+ is required'); + assert.ok(existsSync(bin), 'Build the release CLI first, or set PROBBIT_BIN'); + const profile = mkdtempSync(path.join(out, 'chrome-profile-')); + const chrome = spawn(chromeBinary(), [ + '--headless=new', '--remote-debugging-address=127.0.0.1', '--remote-debugging-port=0', + '--disable-background-networking', '--disable-sync', '--disable-extensions', + '--no-first-run', '--no-default-browser-check', '--user-data-dir=' + profile, 'about:blank', + ]); + children.push(chrome); + let chromeError, chromeStderr = ''; + chrome.on('error', error => { chromeError = error; }); + chrome.stderr.on('data', data => { chromeStderr += data; }); + chrome.stdout.resume(); + servers.push(() => write('chrome-stderr.txt', chromeStderr)); + await until(() => { + if (chromeError) throw chromeError; + if (chrome.exitCode !== null) throw Error('Chrome exited: ' + chromeStderr); + return existsSync(path.join(profile, 'DevToolsActivePort')); + }, 'Chrome DevTools'); + const port = readFileSync(path.join(profile, 'DevToolsActivePort'), 'utf8').split('\n')[0]; + const tabs = await json('http://127.0.0.1:' + port + '/json/list'); + const send = await cdp(tabs.find(tab => tab.type === 'page').webSocketDebuggerUrl); + await send('Page.enable'); + await send('Runtime.enable'); + await send('Emulation.setEmulatedMedia', {features: [{name: 'prefers-reduced-motion', value: 'reduce'}]}); + const value = async expression => { + const result = await send('Runtime.evaluate', {expression, returnByValue: true, awaitPromise: true}); + assert.ok(!result.exceptionDetails, JSON.stringify(result.exceptionDetails)); + return result.result.value; + }; + const viewport = (width, height) => send('Emulation.setDeviceMetricsOverride', {width, height, deviceScaleFactor: 1, mobile: false}); + const navigate = async url => { + await send('Page.navigate', {url}); + await until(() => value(`location.href === ${JSON.stringify(url)} && document.readyState === 'complete'`), 'monitor page load'); + }; + const state = () => value(`({ + event: document.getElementById('ev')?.textContent, + note: document.getElementById('note')?.textContent, + badge: document.getElementById('badge')?.textContent, + line: document.getElementById('line')?.textContent, + why: document.getElementById('why')?.textContent, + saidHidden: document.getElementById('said')?.hidden, + emptyHidden: document.getElementById('empty')?.hidden, + rawHidden: document.getElementById('raw')?.hidden, + drives: document.getElementById('drives')?.hidden, + scrollWidth: document.documentElement.scrollWidth, width: innerWidth, + height: document.documentElement.scrollHeight + })`); + async function shot(name, width, height) { + await viewport(width, height); + await sleep(150); + const frame = await state(); + assert.ok(frame.scrollWidth <= width, 'Horizontal overflow: ' + JSON.stringify(frame)); + const screenshot = await send('Page.captureScreenshot', { + format: 'png', captureBeyondViewport: true, + clip: {x: 0, y: 0, width, height: Math.max(height, frame.height), scale: 1}, + }); + write(name + '.png', Buffer.from(screenshot.data, 'base64')); + report.screenshots.push({name, ...frame}); + } + const event = number => until(async () => (await state()).event === 'event ' + number, 'event ' + number); + const learnError = () => value(`[...document.querySelectorAll('#drives .row')] + .find(row => row.querySelector('.id')?.textContent === 'learn')?.querySelector('.petxt')?.textContent`); + + const demo = await server(['--demo', 'drives'], 'drives'); + await viewport(1440, 1000); + await navigate(demo.url); + await event(3); + report.demo.event3 = await state(); + assert.equal(report.demo.event3.drives, false); + await shot('monitor-drives-desktop', 1440, 1000); + await shot('monitor-drives-narrow', 390, 844); + await event(4); + report.demo.negativeError = await learnError(); + assert.ok(report.demo.negativeError.includes('below expectation')); + await shot('monitor-drives-negative', 1440, 1000); + await event(5); + report.demo.clearedError = await learnError(); + assert.equal(report.demo.clearedError, 'prediction error 0.00'); + await until(async () => (await state()).note?.includes('complete; restarting'), 'demo completion cue'); + report.demo.completed = await state(); + await shot('monitor-drives-completed', 1440, 1000); + await until(async () => (await state()).note?.includes('loop 2'), 'demo loop 2'); + report.demo.restarted = await state(); + assert.equal(report.demo.restarted.event, 'event 0'); + assert.equal(report.demo.restarted.line, ''); + assert.equal(report.demo.restarted.why, ''); + assert.equal(report.demo.restarted.saidHidden, true); + assert.equal(report.demo.restarted.emptyHidden, false); + assert.equal(report.demo.restarted.rawHidden, true); + assert.equal(report.demo.restarted.drives, true); + + const week = await server(['--demo'], 'week'); + await navigate(week.url); + await event(3); + await shot('monitor-week-desktop', 1440, 1000); + await shot('monitor-week-narrow', 390, 844); + + const full = path.join(out, 'follow-full-' + Date.now() + '.strand'); + const growing = path.join(out, 'follow-growing-' + Date.now() + '.strand'); + const outputs = command(['live', 'examples/persona/tutor.yaml', '--seed', '2', '--clock', 'fixed', '--strand', full], + '{"praise":true,"elapsed_hours":1}\n{"loss":true,"elapsed_hours":1}\n{}\n').trim().split('\n').map(JSON.parse); + const lines = readFileSync(full, 'utf8').match(/.*\n/g); + writeFileSync(growing, lines.slice(0, 2).join('')); + const follow = await server([growing], 'follow'); + await viewport(1440, 1000); + await navigate(follow.url); + await event(1); + report.follow.before = await state(); + await shot('monitor-follow-before', 1440, 1000); + appendFileSync(growing, lines[2].slice(0, -1)); + await sleep(350); + assert.equal((await state()).event, 'event 1'); + report.follow.incompleteHeld = true; + const appended = Date.now(); + appendFileSync(growing, '\n'); + await until(async () => (await state()).event === 'event 2', 'appended event', 2000); + report.follow.appendLatencyMs = Date.now() - appended; + assert.ok(report.follow.appendLatencyMs < 1000, 'Follow update exceeded one second'); + report.follow.after = await state(); + assert.equal(report.follow.after.line, outputs[1].line); + assert.deepEqual(await json(follow.url + 'doc/2'), outputs[1]); + await shot('monitor-follow-after', 1440, 1000); + appendFileSync(growing, lines[3]); + await event(3); + assert.ok((await state()).why.includes('prior state still applies')); + assert.equal((await json(follow.url + 'doc/3')).why, 'baseline (no live evidence)'); + report.follow.rawWhyUnchanged = true; + command(['live', 'control', growing, 'pause', '--by', 'human:demo', '--reason', 'synthetic browser check']); + await until(async () => (await state()).badge?.includes('paused'), 'paused status'); + report.follow.paused = await state(); + await shot('monitor-paused-narrow', 390, 844); + + const checkpoint = path.join(out, 'checkpoint-' + Date.now() + '.strand'); + command(['live', 'examples/persona/tutor.yaml', '--seed', '2', '--clock', 'fixed', '--checkpoint-every', '2', '--strand', checkpoint], '{}\n{}\n{}\n'); + const checkpointServer = await server([checkpoint], 'checkpoint'); + await navigate(checkpointServer.url); + await event(3); + report.checkpoint = await state(); + assert.ok(report.checkpoint.badge.includes('from checkpoint 2')); + assert.equal(failures.length, 0, JSON.stringify(failures)); + report.ok = true; +} catch (error) { + report.error = String(error.stack || error); + process.exitCode = 1; +} finally { + for (const save of servers) save(); + write('browser-report.json', JSON.stringify(report, null, 2) + '\n'); + for (const ws of sockets) ws.close(); + for (const child of children) child.kill('SIGTERM'); + console.log(JSON.stringify({artifacts: out, ...report}, null, 2)); +} diff --git a/probbit-cli/tests/processor_contracts.rs b/probbit-cli/tests/processor_contracts.rs new file mode 100644 index 0000000..653841e --- /dev/null +++ b/probbit-cli/tests/processor_contracts.rs @@ -0,0 +1,207 @@ +//! Public processor responses: released assignments are never confused with the +//! legacy full diagnostic candidate. Fixed-work behavior is thread invariant. +#[allow(dead_code)] +#[path = "../src/json.rs"] +mod json; +use json::Json; +use std::io::Write; +use std::process::{Command, Stdio}; + +fn call(args: &[&str], input: &str) -> (i32, Json) { + let mut c = Command::new(env!("CARGO_BIN_EXE_probbit")) + .args(args) + .env_remove("PROBBIT_CONFIG") + .env_remove("PROBBIT_THREADS") + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .spawn() + .unwrap(); + c.stdin.take().unwrap().write_all(input.as_bytes()).unwrap(); + let o = c.wait_with_output().unwrap(); + let s = String::from_utf8(o.stdout).unwrap(); + ( + o.status.code().unwrap(), + json::parse(&s).unwrap_or_else(|_| { + panic!( + "invalid output {s}; stderr {}", + String::from_utf8_lossy(&o.stderr) + ) + }), + ) +} +fn contract(j: &Json) { + let verdict = j.get("verdict").and_then(Json::as_str).unwrap(); + let Some(plan) = j.get("plan") else { + assert!(["infeasible", "refused", "declined"].contains(&verdict)); + assert!(j.get("released_plan").is_none()); + return; + }; + let expected = match verdict { + "exact" | "diagnostics_passed" => "released", + "partial" => "partial", + _ => "diagnostic", + }; + assert_eq!(j.get("plan_status").and_then(Json::as_str), Some(expected)); + let released = j.get("released").and_then(Json::as_arr).unwrap(); + let rp = j.get("released_plan").and_then(Json::as_obj).unwrap(); + assert_eq!(rp.len(), released.len()); + for (id, value) in rp { + assert!(released.contains(&json::str(id))); + assert_eq!(plan.get(id), Some(value)); + } + let count = plan.as_obj().unwrap().len(); + match expected { + "released" => assert_eq!(rp.len(), count), + "partial" => assert!(!rp.is_empty() && rp.len() < count), + _ => assert!(rp.is_empty()), + } + if let Some(answers) = j.get("answers").and_then(Json::as_obj) { + for (id, a) in answers { + assert_eq!( + a.get("probbit").unwrap().get("released"), + Some(&Json::Bool(released.contains(&json::str(id)))) + ); + } + } +} +fn stable(j: &Json) -> Json { + match j { + Json::Obj(v) => Json::Obj( + v.iter() + .filter(|(k, _)| { + ![ + "ms", + "sample_ms", + "gate_ms", + "site_updates_per_s", + "process_cpu_ms", + "peak_rss_mb", + "nice", + "phases", + "threads", + ] + .contains(&k.as_str()) + }) + .map(|(k, v)| (k.clone(), stable(v))) + .collect(), + ), + Json::Arr(v) => Json::Arr(v.iter().map(stable).collect()), + x => x.clone(), + } +} + +#[test] +fn exact_refused_and_infeasible_plans_have_distinct_public_contracts() { + let doc = r#"{"probbit_ir":1,"values":["a","b"],"vars":[{"id":"x"},{"id":"y"}]}"#; + for args in [ + vec!["run"], + vec!["run", "--op", "sample", "--sweeps", "1", "--polish-ms", "0"], + vec!["run", "--op", "exact", "--exact-ms", "0"], + ] { + let (_, j) = call(&args, doc); + contract(&j); + } + let (_, bad) = call( + &["run"], + r#"{"probbit_ir":1,"values":["a"],"vars":[{"id":"x"}],"caps":[{"value":"a","limit":0}]}"#, + ); + assert_eq!( + bad.get("verdict").and_then(Json::as_str), + Some("infeasible") + ); + contract(&bad); + let (_, d) = call(&["demo", "--tasks", "12"], ""); + let d = json::write(&d, false); + for args in [ + vec!["decide"], + vec![ + "decide", + "--mode", + "sample", + "--sweeps", + "1", + "--polish-ms", + "0", + ], + ] { + contract(&call(&args, &d).1); + } + let ev = include_str!("../../examples/evaluate/support-12.json"); + for args in [ + vec!["evaluate"], + vec![ + "evaluate", + "--op", + "sample", + "--sweeps", + "1", + "--polish-ms", + "0", + ], + ] { + contract(&call(&args, ev).1); + } +} + +#[test] +fn fixed_work_release_and_candidate_fields_match_at_one_two_and_four_threads() { + let (_, d) = call(&["demo", "--tasks", "300"], ""); + let d = json::write(&d, false); + let mut partial = 0; + for (cmd, input) in [ + ("decide", d.as_str()), + ("run", include_str!("../../examples/knapsack-20.json")), + ( + "evaluate", + include_str!("../../examples/evaluate/support-12.json"), + ), + ] { + let mut first = None; + for threads in ["1", "2", "4"] { + let (code, j) = call( + &[ + cmd, + if cmd == "decide" { "--mode" } else { "--op" }, + "sample", + "--sweeps", + "1000", + "--polish-sweeps", + "50", + "--threads", + threads, + ], + input, + ); + contract(&j); + if j.get("verdict").and_then(Json::as_str) == Some("partial") { + partial += 1; + } + let got = (code, stable(&j)); + if let Some(expected) = &first { + assert_eq!(&got, expected, "{cmd} threads {threads}"); + } else { + first = Some(got); + } + } + } + assert!( + partial >= 3, + "the fixture must exercise a real partial release" + ); +} + +#[test] +fn malformed_input_is_one_error_document_not_a_panic() { + for input in [ + r#"{"probbit_ir":01}"#, + "{\"probbit_ir\":1,\"comment\":\"a\nb\"}", + r#"{"probbit_ir":1,"values":["a"],"vars":[{"h":{"a":1e309}}]}"#, + r#"{"probbit_ir":1,"values":["a","b"],"vars":[{"id":"x"},{"id":"y"}],"pairs":[{"i":"x","j":"y","table":[[0,0]]}]}"#, + ] { + let (code, j) = call(&["run"], input); + assert_eq!(code, 2); + assert_eq!(j.as_obj().unwrap().len(), 1); + assert!(j.get("error").is_some()); + } +} diff --git a/probbit-cli/tests/processor_json.rs b/probbit-cli/tests/processor_json.rs new file mode 100644 index 0000000..4833f85 --- /dev/null +++ b/probbit-cli/tests/processor_json.rs @@ -0,0 +1,46 @@ +#[allow(dead_code)] +#[path = "../src/json.rs"] +mod json; + +#[test] +fn json_numbers_follow_the_grammar_not_rust_float_syntax() { + for bad in [ + "01", "-01", "1.", "-.1", "00", "1.e2", "1e", "1e+", "+1", ".1", "--1", "1e+-2", + ] { + assert!( + json::parse(bad).is_err(), + "accepted invalid JSON number {bad}" + ); + assert!( + json::parse(&format!("{{\"x\":{bad}}}")).is_err(), + "accepted nested {bad}" + ); + } + for good in [ + "0", + "-0", + "1.0", + "-0.1", + "1e2", + "1E+2", + "1e-2", + "9007199254740992", + "1e-300", + ] { + assert!(json::parse(good).is_ok(), "rejected {good}"); + } +} + +#[test] +fn json_strings_reject_all_unescaped_control_characters() { + for c in 0..32u8 { + assert!( + json::parse(&format!("\"a{}b\"", c as char)).is_err(), + "accepted control {c}" + ); + assert!(json::parse(&format!("\"a\\u{c:04x}b\"")).is_ok()); + } + for good in [r#""😀""#, r#""\ud83d\ude00""#, r#""\n\t\r\b\f\/\\\"""#] { + assert!(json::parse(good).is_ok()); + } +} diff --git a/probbit-core/tests/rng_contract.rs b/probbit-core/tests/rng_contract.rs new file mode 100644 index 0000000..016649e --- /dev/null +++ b/probbit-core/tests/rng_contract.rs @@ -0,0 +1,38 @@ +use probbit_core::{Philox4x32, SplitMix64}; + +#[test] +fn philox_matches_published_random123_zero_vector() { + // Random123's Philox4x32-10 known-answer test: counter and key all zero. + let mut rng = Philox4x32::new(0, 0); + assert_eq!( + [ + rng.next_u32(), + rng.next_u32(), + rng.next_u32(), + rng.next_u32() + ], + [0x6627e8d5, 0xe169c58d, 0xbc57ac4c, 0x9b00dbd8] + ); +} + +#[test] +fn splitmix_and_philox_stream_contracts_are_pinned() { + let mut r = SplitMix64(0); + assert_eq!(r.next_u64(), 0xe220a8397b1dcdaf); + assert_eq!(r.next_u64(), 0x6e789e6aa1b965f4); + let mut whole = Philox4x32::new(7, 11); + let mut words = whole.clone(); + for _ in 0..1000 { + assert_eq!( + whole.next_u64(), + ((words.next_u32() as u64) << 32) | words.next_u32() as u64 + ); + } + let mut a = Philox4x32::new(13, 19); + let mut b = a.clone(); + for _ in 0..1000 { + let x = a.f64(); + assert!((0.0..1.0).contains(&x)); + assert_eq!(x, (b.next_u64() >> 11) as f64 / 9007199254740992.0); + } +} diff --git a/probbit-decide/src/lib.rs b/probbit-decide/src/lib.rs index 7cd5392..088c65f 100644 --- a/probbit-decide/src/lib.rs +++ b/probbit-decide/src/lib.rs @@ -937,6 +937,7 @@ pub fn polish_plan(p: &Problem, start: Option<&[usize]>, ms: f64, seed: u64) -> pub fn polish_plan_sweeps(p: &Problem, start: Option<&[usize]>, sweeps: usize, seed: u64) -> Option<(f64, Vec)> { polish_plan_on(p, start, 0.0, sweeps.max(1), 4, seed) } /// The polish's 4 chains on `threads` workers (`--threads`); see `probbit_ir::anneal_on` for the ms / sweeps semantics. pub fn polish_plan_on(p: &Problem, start: Option<&[usize]>, ms: f64, sweeps: usize, threads: usize, seed: u64) -> Option<(f64, Vec)> { + if start.is_some_and(|x| x.len() != p.t || x.iter().any(|&a| a >= p.a) || p.violations(x) != 0) { return None; } let seq = probbit_core::rt::sequential(); let t = if seq { 1 } else { threads.clamp(1, 4) }; let ms = ms / 4usize.div_ceil(t) as f64; let betas = [2.0, 4.0, 8.0, 16.0, 32.0]; let qs: Vec = betas.iter().map(|&b| { let mut q = p.clone(); for v in q.h.iter_mut() { *v *= b; } q.lam *= b; q }).collect(); @@ -951,7 +952,7 @@ pub fn polish_plan_on(p: &Problem, start: Option<&[usize]>, ms: f64, sweeps: usi if sweeps == 0 && t0.elapsed().as_secs_f64() * 1e3 >= ms { break; } let mut ch = Chain::new(q, seed ^ 0x5eed, (c * 8 + k) as u64)?; ch.x = x.clone(); ch.load = vec![0; p.a]; for &a in &x { ch.load[a] += 1; } let stop = ms * (k + 1) as f64 / betas.len() as f64; let mut j = 0usize; - let n = sweeps * (k + 1) / betas.len() - sweeps * k / betas.len(); + let n = ((sweeps as u128 * (k + 1) as u128) / betas.len() as u128 - (sweeps as u128 * k as u128) / betas.len() as u128) as usize; while if sweeps > 0 { j < n } else { j % 4 != 0 || t0.elapsed().as_secs_f64() * 1e3 < stop } { ch.sweep(); j += 1; let lw = p.logw(&ch.x); if lw > best.0 { best = (lw, ch.x.clone()); } } x = ch.x.clone(); } diff --git a/probbit-decide/tests/correctness.rs b/probbit-decide/tests/correctness.rs new file mode 100644 index 0000000..3818563 --- /dev/null +++ b/probbit-decide/tests/correctness.rs @@ -0,0 +1,111 @@ +use probbit_decide::{exact, polish_plan_on, Problem}; + +fn fixture(case: usize) -> Problem { + let (t, a) = (1 + case % 6, 2 + case % 3); + Problem { + t, + a, + h: (0..t * a) + .map(|i| ((i * 17 + case * 7) % 29) as f64 / 7.0 - 2.0) + .collect(), + allowed: (0..t * a).map(|i| (i * 11 + case) % 7 != 0).collect(), + cap: (0..a).map(|i| (case + i * 3) % (t + 1)).collect(), + group: (0..t) + .map(|i| if i % 3 == 0 { usize::MAX } else { i % 2 }) + .collect(), + lam: (case % 5) as f64 / 3.0, + clamp: (0..t) + .map(|i| { + if (case + i) % 11 == 0 { + Some(case % a) + } else { + None + } + }) + .collect(), + block_moves: false, + pair_swaps: false, + collective: true, + cluster: true, + cycles: true, + } +} +fn close(a: f64, b: f64) { + assert!((a - b).abs() < 2e-10, "{a} != {b}"); +} + +#[test] +fn router_and_lowering_match_180_independent_exhaustive_oracles() { + for case in 0..180 { + let p = fixture(case); + let mut rows = vec![]; + for mut code in 0..p.a.pow(p.t as u32) { + let x: Vec<_> = (0..p.t) + .map(|_| { + let a = code % p.a; + code /= p.a; + a + }) + .collect(); + if (0..p.t).any(|i| !p.allowed[i * p.a + x[i]] || p.clamp[i].is_some_and(|a| a != x[i])) + { + continue; + } + if (0..p.a).any(|a| x.iter().filter(|&&v| v == a).count() > p.cap[a]) { + continue; + } + let unary: f64 = x.iter().enumerate().map(|(i, &a)| p.h[i * p.a + a]).sum(); + let mut same = 0; + for i in 0..p.t { + for j in i + 1..p.t { + if p.group[i] != usize::MAX && p.group[i] == p.group[j] && x[i] == x[j] { + same += 1; + } + } + } + rows.push((unary + p.lam * same as f64, x)); + } + let e = exact(&p, 5, 100_000).unwrap(); + let lowered = probbit_ir::exact(&p.lower(), 5, 100_000).unwrap(); + assert_eq!(e.n_feasible as usize, rows.len()); + assert_eq!(lowered.n_feasible, e.n_feasible); + if rows.is_empty() { + assert!(e.top.is_empty()); + assert_eq!(e.logz, f64::NEG_INFINITY); + continue; + } + let best = rows + .iter() + .map(|(w, _)| *w) + .fold(f64::NEG_INFINITY, f64::max); + let z: f64 = rows.iter().map(|(w, _)| (w - best).exp()).sum(); + let logz = best + z.ln(); + close(e.logz, logz); + close(lowered.logz, logz); + close(p.logw(&e.top[0].1), best); + let mut marg = vec![0.0; p.t * p.a]; + for (w, x) in rows { + for (i, a) in x.into_iter().enumerate() { + marg[i * p.a + a] += (w - best).exp() / z; + } + } + for (i, &want) in marg.iter().enumerate() { + close(e.marg[i], want); + close(lowered.marg[i], want); + } + } +} + +#[test] +fn router_polish_refuses_infeasible_or_malformed_starts() { + let mut p = fixture(1); + p.allowed.fill(true); + p.clamp.fill(None); + p.cap = vec![1; p.a]; + for start in [vec![0, 0], vec![], vec![0], vec![0, p.a], vec![0, 1, 0]] { + assert!( + polish_plan_on(&p, Some(&start), 0.0, 0, 1, 7).is_none(), + "accepted {start:?}" + ); + } +} diff --git a/probbit-ir/src/lib.rs b/probbit-ir/src/lib.rs index 619f082..6b340d2 100644 --- a/probbit-ir/src/lib.rs +++ b/probbit-ir/src/lib.rs @@ -71,13 +71,20 @@ const NONE: u32 = u32::MAX; impl Model { pub fn new(n: usize, k: usize, h: Vec, allowed: Vec, clamp: Vec>, pairs: Vec, caps: Vec) -> Result { - if h.len() != n * k || allowed.len() != n * k || clamp.len() != n { return Err("h / allowed must have n*k entries and clamp n".into()); } if k == 0 || k > 65535 { return Err("k must be in 1..=65535".into()); } + let nk = n.checked_mul(k).ok_or("n*k exceeds the platform size limit")?; + if h.len() != nk || allowed.len() != nk || clamp.len() != n { return Err("h / allowed must have n*k entries and clamp n".into()); } + if h.iter().any(|x| !x.is_finite()) { return Err("h must contain only finite weights".into()); } if let Some(i) = clamp.iter().position(|c| c.map_or(false, |v| v >= k)) { return Err(format!("clamp of variable {i} out of range")); } let mut adj = vec![vec![]; n]; for (q, p) in pairs.iter().enumerate() { if p.i >= n || p.j >= n || p.i == p.j { return Err(format!("pair {q}: bad variables ({}, {})", p.i, p.j)); } - if let Coupling::Table(t) = &p.c { if t.len() != k * k { return Err(format!("pair {q}: table must have k*k entries")); } } + match &p.c { + Coupling::Table(t) => { if t.len() != k * k { return Err(format!("pair {q}: table must have k*k entries")); } + if t.iter().any(|x| !x.is_finite()) { return Err(format!("pair {q}: table must contain only finite weights")); } } + Coupling::Potts(w) if !w.is_finite() => return Err(format!("pair {q}: potts weight must be finite")), + _ => {} + } adj[p.i].push((p.j, q)); adj[p.j].push((p.i, q)); } for a in adj.iter_mut() { a.sort(); } @@ -85,6 +92,11 @@ impl Model { for (c, cp) in caps.iter().enumerate() { if !cp.weights.is_empty() && cp.weights.len() != cp.members.len() { return Err(format!("capacity {c}: {} weights for {} members", cp.weights.len(), cp.members.len())); } if let Some(&w) = cp.weights.iter().find(|&&w| w == 0 || w > MAX_CAP_WEIGHT) { return Err(format!("capacity {c}: weight {w} outside 1..={MAX_CAP_WEIGHT}")); } + // Loads are usize on both native and wasm32. Reject an unrepresentable + // bound before any sampler/search can wrap a load and admit an invalid plan. + if cp.weights.iter().try_fold(0usize, |s, &w| s.checked_add(w)).is_none() { + return Err(format!("capacity {c}: total member weight exceeds the platform size limit")); + } for &(i, v) in &cp.members { if i >= n || v >= k { return Err(format!("capacity {c}: member ({i}, {v}) out of range")); } if cap_of[i * k + v].last() == Some(&c) { return Err(format!("capacity {c}: duplicate member ({i}, {v})")); } @@ -574,7 +586,7 @@ pub fn compile_parts(k: usize, h: &mut [f64], pairs: Vec, caps: Vec) let nc = caps.len(); // never binding: the most one plan can load the cap (per variable its heaviest member; unit = its distinct member variables) fits let caps: Vec = caps.into_iter().filter(|cp| { let mut v: Vec<(usize, usize)> = cp.members.iter().enumerate().map(|(t, &(i, _))| (i, cp.w(t))).collect(); v.sort_unstable(); - let most: usize = v.iter().enumerate().filter(|&(q, &(i, _))| v.get(q + 1).map_or(true, |&(j, _)| j != i)).map(|(_, &(_, w))| w).sum(); cp.limit < most }).collect(); + let most: u128 = v.iter().enumerate().filter(|&(q, &(i, _))| v.get(q + 1).map_or(true, |&(j, _)| j != i)).map(|(_, &(_, w))| w as u128).sum(); (cp.limit as u128) < most }).collect(); c.caps_dropped = nc - caps.len(); (kept, caps, c) } @@ -1040,6 +1052,7 @@ pub fn anneal_sweeps(m: &Model, start: Option<&[usize]>, betas: &[f64], sweeps: /// chain gets ms / ceil(chains / threads), as in the sampler. threads >= chains = the old one-thread-per-chain behaviour. #[allow(clippy::too_many_arguments)] pub fn anneal_on(m: &Model, start: Option<&[usize]>, betas: &[f64], ms: f64, sweeps: usize, chains: usize, threads: usize, seed: u64) -> Option<(f64, Vec)> { + if start.is_some_and(|x| x.len() != m.n || x.iter().any(|&v| v >= m.k) || m.violations(x) != 0) { return None; } let seq = probbit_core::rt::sequential(); let t = if seq { 1 } else { threads.clamp(1, chains.max(1)) }; let ms = ms / chains.div_ceil(t) as f64; // R19.7: one model, the stage's beta on the chain (was a scaled Model clone per beta, built before the clock started) let run = |c: usize| -> Option<(f64, Vec)> { @@ -1048,7 +1061,7 @@ pub fn anneal_on(m: &Model, start: Option<&[usize]>, betas: &[f64], ms: f64, swe for (q, &b) in betas.iter().enumerate() { let mut ch = Chain::from_state(m, seed ^ 0x5eed, (c * 64 + q) as u64, &x); ch.beta = b; ch.plain = true; let stop = ms * (q + 1) as f64 / nq as f64; let mut j = 0usize; - let n = sweeps * (q + 1) / nq - sweeps * q / nq; + let n = ((sweeps as u128 * (q + 1) as u128) / nq as u128 - (sweeps as u128 * q as u128) / nq as u128) as usize; while if sweeps > 0 { j < n } else { j % 4 != 0 || t0.elapsed().as_secs_f64() * 1e3 < stop } { ch.sweep(); j += 1; let lw = m.logw(&ch.x); if lw > best.0 { best = (lw, ch.x.clone()); } } x = ch.x.clone(); } diff --git a/probbit-ir/tests/correctness.rs b/probbit-ir/tests/correctness.rs new file mode 100644 index 0000000..2225b4f --- /dev/null +++ b/probbit-ir/tests/correctness.rs @@ -0,0 +1,234 @@ +//! Independent exhaustive checks: the oracle does not call Model::logw, violations, +//! candidate/start search, compilation, or any exact solver. +use probbit_ir::{Cap, Coupling, Model, Pair}; + +struct Rng(u64); +impl Rng { + fn next(&mut self, n: usize) -> usize { + self.0 = self + .0 + .wrapping_mul(6364136223846793005) + .wrapping_add(1442695040888963407); + ((self.0 >> 32) % n as u64) as usize + } + fn weight(&mut self) -> f64 { + self.next(17) as f64 / 4.0 - 2.0 + } +} + +fn feasible(m: &Model, x: &[usize]) -> bool { + (0..m.n).all(|i| m.allowed[i * m.k + x[i]] && m.clamp[i].map_or(true, |v| v == x[i])) + && m.caps.iter().all(|c| { + c.members + .iter() + .enumerate() + .filter(|(_, (i, v))| x[*i] == *v) + .map(|(q, _)| c.weights.get(q).copied().unwrap_or(1) as u64) + .sum::() + <= c.limit as u64 + }) +} +fn score(m: &Model, x: &[usize]) -> f64 { + let mut w = (0..m.n).map(|i| m.h[i * m.k + x[i]]).sum::(); + for p in &m.pairs { + w += match &p.c { + Coupling::Potts(v) => { + if x[p.i] == x[p.j] { + *v + } else { + 0.0 + } + } + Coupling::Table(t) => t[x[p.i] * m.k + x[p.j]], + }; + } + w +} +fn oracle(m: &Model) -> (usize, f64, f64, Vec) { + let mut states = vec![]; + for mut code in 0..m.k.pow(m.n as u32) { + let x: Vec<_> = (0..m.n) + .map(|_| { + let v = code % m.k; + code /= m.k; + v + }) + .collect(); + if feasible(m, &x) { + states.push((score(m, &x), x)); + } + } + let best = states + .iter() + .map(|(w, _)| *w) + .fold(f64::NEG_INFINITY, f64::max); + let z: f64 = states.iter().map(|(w, _)| (w - best).exp()).sum(); + let mut marg = vec![0.0; m.n * m.k]; + for (w, x) in &states { + for (i, &v) in x.iter().enumerate() { + marg[i * m.k + v] += (w - best).exp() / z; + } + } + (states.len(), best + z.ln(), best, marg) +} +fn close(a: f64, b: f64) { + assert!((a - b).abs() < 2e-10, "{a} != {b}"); +} + +#[test] +fn exact_tiers_and_compilation_match_320_independent_exhaustive_oracles() { + let mut r = Rng(0x534d414c4c); + let mut answered_components = 0; + let mut answered_frontier = 0; + for case in 0..320 { + let (n, k) = (1 + r.next(6), 2 + r.next(3)); + let h = (0..n * k).map(|_| r.weight()).collect(); + let allowed = (0..n * k).map(|_| r.next(7) != 0).collect(); + let clamp = (0..n) + .map(|_| { + if r.next(5) == 0 { + Some(r.next(k)) + } else { + None + } + }) + .collect(); + let mut pairs = vec![]; + for i in 0..n { + for j in i + 1..n { + if r.next(4) == 0 { + // Include reverse-oriented, asymmetric tables; non-negative and negative Potts terms. + pairs.push(if r.next(2) == 0 { + Pair { + i, + j, + c: Coupling::Potts(r.weight()), + } + } else { + Pair { + i: j, + j: i, + c: Coupling::Table((0..k * k).map(|_| r.weight()).collect()), + } + }); + } + } + } + let mut caps = vec![]; + for _ in 0..r.next(4) { + let members: Vec<_> = (0..n) + .flat_map(|i| (0..k).map(move |v| (i, v))) + .filter(|_| r.next(4) == 0) + .collect(); + let weights = if case % 3 == 0 { + members.iter().map(|_| 1 + r.next(4)).collect() + } else { + vec![] + }; + caps.push(Cap { + members, + weights, + limit: r.next(n + 3), + }); + } + let m = Model::new(n, k, h, allowed, clamp, pairs, caps).unwrap(); + let (count, logz, best, marg) = oracle(&m); + for actual in [&m, &m.compiled().0] { + let e = probbit_ir::exact(actual, 5, 100_000).unwrap(); + assert_eq!(e.n_feasible as usize, count, "case {case}"); + if count == 0 { + assert!(e.top.is_empty()); + assert_eq!(e.logz, f64::NEG_INFINITY); + } else { + close(e.logz, logz); + close(score(&m, &e.top[0].1), best); + for (a, b) in e.marg.iter().zip(&marg) { + close(*a, *b); + } + for (p, x) in &e.top { + assert!(feasible(&m, x)); + close(*p, (score(&m, x) - logz).exp()); + } + } + if let Some(c) = probbit_ir::exact_components_until(actual, 100_000, 100_000, None) { + answered_components += 1; + assert_eq!(c.infeasible, count == 0, "case {case}"); + if count > 0 { + close(c.logz, logz); + close(c.map_logw, best); + assert!(feasible(&m, &c.map)); + for (a, b) in c.marg.iter().zip(&marg) { + close(*a, *b); + } + } + } + if let Some(f) = probbit_ir::exact_frontier(actual, 100_000) { + answered_frontier += 1; + assert!(count > 0); + close(f.logz, logz); + close(f.map_logw, best); + assert!(feasible(&m, &f.map)); + for (a, b) in f.marg.iter().zip(&marg) { + close(*a, *b); + } + } + } + } + assert!(answered_components > 100 && answered_frontier > 100); + eprintln!("320 independent oracles; 640 enumerations; {answered_components} component answers; {answered_frontier} frontier answers"); +} + +#[test] +fn constructors_reject_overflow_and_non_finite_weights() { + assert!(Model::new(usize::MAX, 2, vec![], vec![], vec![], vec![], vec![]).is_err()); + for bad in [f64::NAN, f64::INFINITY, f64::NEG_INFINITY] { + assert!(Model::new(1, 1, vec![bad], vec![true], vec![None], vec![], vec![]).is_err()); + for c in [Coupling::Potts(bad), Coupling::Table(vec![bad])] { + assert!(Model::new( + 2, + 1, + vec![0.0; 2], + vec![true; 2], + vec![None; 2], + vec![Pair { i: 0, j: 1, c }], + vec![] + ) + .is_err()); + } + } + assert!(Model::new(1, 2, vec![0.0], vec![true; 2], vec![None], vec![], vec![]).is_err()); + assert!(Model::new( + 2, + 2, + vec![0.0; 4], + vec![true; 4], + vec![None; 2], + vec![Pair { + i: 0, + j: 1, + c: Coupling::Table(vec![0.0; 3]) + }], + vec![] + ) + .is_err()); +} + +#[test] +fn polish_never_returns_an_invalid_supplied_start() { + let m = Model::new( + 2, + 2, + vec![0.0; 4], + vec![true; 4], + vec![None; 2], + vec![], + vec![Cap::new(vec![(0, 0), (1, 0)], 1)], + ) + .unwrap(); + for start in [vec![0, 0], vec![], vec![0], vec![0, 2], vec![0, 1, 0]] { + assert!( + probbit_ir::anneal_on(&m, Some(&start), &[1.0, 2.0], 0.0, 0, 1, 1, 7).is_none(), + "accepted {start:?}" + ); + } +} diff --git a/probbit-wasm/src/lib.rs b/probbit-wasm/src/lib.rs index 0a49540..c6d3263 100644 --- a/probbit-wasm/src/lib.rs +++ b/probbit-wasm/src/lib.rs @@ -68,7 +68,9 @@ pub fn demo(input: &str) -> (String, i32) { let int = |k: &str, d: usize| o.iter().find(|(x, _)| x == k).map_or(Ok(d), |(_, v)| json::count(v, k).map_err(|e| e.to_json())); let hard = match o.iter().find(|(x, _)| x == "hard") { None => false, Some((_, Json::Bool(b))) => *b, Some(_) => return Err(json::schema("hard", "must be true or false").to_json()) }; let tasks = int("tasks", 12)?; if tasks == 0 { return Err(json::value("tasks", "at least 1").to_json()); } - Ok(json::write(&router::demo_doc(tasks, int("seed", 7)? as u64, hard), true)) + if tasks > json::MAX_DENSE / 6 { return Err(json::limit("tasks", format!("at most {} tasks for the six-worker demo", json::MAX_DENSE / 6)).to_json()); } + let seed = o.iter().find(|(k, _)| k == "seed").map_or(Ok(7), |(_, v)| json::count_u64(v, "seed").map_err(|e| e.to_json()))?; + Ok(json::write(&router::demo_doc(tasks, seed, hard), true)) }; match go() { Ok(s) => (s, 0), Err(e) => (json::write(&e, false), 2) } } @@ -95,30 +97,31 @@ fn flags(f: Option<&Json>, cmd: &str) -> Result { let mut o = Flags { budget: 200.0, budget_given: false, seed: 7, exact_limit: 2_000_000, exact_ms: None, fr_states: probbit_ir::FRONTIER_MAX_STATES, polish_ms: 50.0, polish_sweeps: 0, sweeps: 0, collective: true, cluster: true, cycles: true, chains: 4, threads: 1, mem_mb: 1024, mode: if cmd == "decide" { "auto" } else { "decide" }.to_string(), deadline_ms: None, program: false }; - let kv = match f { None | Some(Json::Null) => return Ok(o), Some(Json::Obj(v)) => v, Some(_) => return Err(flag_err("", "\"flags\" must be an object")) }; + let kv = match f { None | Some(Json::Null) => return Ok(o), Some(j @ Json::Obj(_)) => json::keyed(j, "flags").map_err(|e| flag_err(e.path.strip_prefix("flags.").unwrap_or(""), &e.msg))?, Some(_) => return Err(flag_err("", "\"flags\" must be an object")) }; let ms = |k: &str, x: &Json, lo_open: bool| match x.as_f64() { Some(v) if v.is_finite() && v <= 1e9 && (v > 0.0 || (!lo_open && v == 0.0)) => Ok(v), _ => Err(flag_err(k, if lo_open { "milliseconds above 0, at most 1e9" } else { "milliseconds from 0 to 1e9" })) }; let int = |k: &str, x: &Json, lo: u64, hi: u64| match x.as_f64() { Some(v) if v.fract() == 0.0 && v >= lo as f64 && v <= hi as f64 => Ok(v as u64), _ => Err(flag_err(k, &format!("an integer from {lo} to {hi}"))) }; let on = |k: &str, x: &Json| match x { Json::Bool(b) => Ok(*b), _ => Err(flag_err(k, "true or false")) }; let max = 1u64 << 53; + let size_max = max.min(usize::MAX as u64); for (k, x) in kv { match (k.as_str(), cmd) { ("budget_ms", _) => { o.budget = ms(k, x, false)?; o.budget_given = true; } ("seed", _) => o.seed = int(k, x, 0, max)?, ("exact_limit", _) => o.exact_limit = int(k, x, 0, max)?, ("exact_ms", _) => o.exact_ms = Some(ms(k, x, false)?), - ("frontier_states", _) => o.fr_states = int(k, x, 0, max)? as usize, + ("frontier_states", _) => o.fr_states = int(k, x, 0, size_max)? as usize, ("polish_ms", _) => o.polish_ms = ms(k, x, false)?, - ("polish_sweeps", _) => o.polish_sweeps = int(k, x, 0, max)? as usize, - ("sweeps", _) => o.sweeps = int(k, x, 0, max)? as usize, + ("polish_sweeps", _) => o.polish_sweeps = int(k, x, 0, size_max)? as usize, + ("sweeps", _) => o.sweeps = int(k, x, 0, size_max)? as usize, ("collective", _) => o.collective = on(k, x)?, ("cluster", _) => o.cluster = on(k, x)?, ("cycles", _) => o.cycles = on(k, x)?, ("chains", _) => o.chains = int(k, x, 1, 100_000)? as usize, ("threads", _) => { o.threads = int(k, x, 1, 1024)? as usize; if cfg!(all(target_family = "wasm", target_os = "unknown")) && o.threads > 1 { return Err(flag_err(k, "this build has no threads (wasm32-unknown-unknown): 1")); } } - ("mem_limit_mb", _) => o.mem_mb = int(k, x, 0, max)? as usize, + ("mem_limit_mb", _) => o.mem_mb = int(k, x, 0, size_max)? as usize, ("mode", "decide") => { o.mode = x.as_str().filter(|m| ["auto", "exact", "sample"].contains(m)).ok_or_else(|| flag_err(k, "auto, exact or sample"))?.to_string(); } ("op", "run" | "evaluate") => { o.mode = x.as_str().filter(|m| ["decide", "exact", "sample"].contains(m)).ok_or_else(|| flag_err(k, "decide, exact or sample"))?.to_string(); } ("deadline_ms", "run" | "evaluate") => o.deadline_ms = Some(ms(k, x, true)?), diff --git a/probbit-wasm/tests/boundaries.mjs b/probbit-wasm/tests/boundaries.mjs new file mode 100644 index 0000000..2f56a94 --- /dev/null +++ b/probbit-wasm/tests/boundaries.mjs @@ -0,0 +1,108 @@ +// Actual wasm32 boundary regressions and CLI parity. No npm packages. +// node probbit-wasm/tests/boundaries.mjs [module.wasm] [native-probbit] +import { readFileSync } from 'node:fs'; +import { strict as assert } from 'node:assert'; +import { spawnSync } from 'node:child_process'; +const modulePath = process.argv[2] || new URL('../../target/wasm32-unknown-unknown/release/probbit_wasm.wasm', import.meta.url); +const native = process.argv[3]; +const { instance: { exports: e } } = await WebAssembly.instantiate(readFileSync(modulePath), { probbit: { now_ms: () => performance.now() } }); +let calls = 0; +function call(op, input) { + const b = typeof input === 'string' ? new TextEncoder().encode(input) : input; + const p = e.probbit_alloc(b.length); + new Uint8Array(e.memory.buffer, p, b.length).set(b); + const n = e.probbit_call(op, p, b.length); + const out = JSON.parse(new TextDecoder().decode(new Uint8Array(e.memory.buffer, e.probbit_out_ptr(), n))); + calls++; + return { code: e.probbit_out_code(), out }; +} +const run = j => call(1, JSON.stringify(j)); +const base = { probbit_ir: 1, values: ['a', 'b'], vars: [{ id: 'x', clamp: 'b' }, { id: 'y', clamp: 'b' }] }; +function error(result, code = 2) { + assert.equal(result.code, code); + assert.deepEqual(Object.keys(result.out), ['error']); + assert.ok(result.out.error.code && typeof result.out.error.message === 'string'); +} +// A u64-to-usize cast used to silently turn 2^32 fixed sweeps into zero; +// capacities saturated, changing the accepted request on 32-bit targets. +for (const flag of ['sweeps', 'polish_sweeps', 'frontier_states', 'mem_limit_mb']) { + const r = run({ ...base, flags: { [flag]: 2 ** 32, budget_ms: 0, polish_ms: 0 } }); + error(r); assert.equal(r.out.error.path, 'flags.' + flag); +} +error(run({ ...base, caps: [{ value: 'a', limit: 2 ** 32 }] })); +error(call(1, '{"flags":{"sweeps":1,"sweeps":2}}')); +for (const text of ['01', '-.1', '1.', '1.e2', '"raw\nnewline"', '1e309']) error(call(1, text)); +error(call(1, new Uint8Array([0xff]))); +error(call(99, '{}')); +error(call(1, '')); +error(call(3, '{"tasks":3333334}')); +// The seed is genuinely u64, not a usize; the larger seed must survive wasm32. +const seed64 = call(3, '{"tasks":1,"seed":4294967296}'); assert.equal(seed64.code, 0); +const seed32 = call(3, '{"tasks":1,"seed":4294967295}'); assert.notDeepEqual(seed64.out, seed32.out); + +// 1 + u32::MAX wrapped to zero, turning impossible precedence into exact success. +const precedence = { ...base, precedes: [{ before: 'x', after: 'y', gap: 4294967295 }] }; +const gap = run(precedence); assert.equal(gap.code, 1); assert.equal(gap.out.verdict, 'infeasible'); assert.equal(gap.out.plan, undefined); + +// 4,295 * 1,000,000 wrapped to 32,704, so compile_parts dropped a binding +// capacity whose limit was 5,000,000. Never emit an exact answer after that loss. +const weighted = { probbit_ir: 1, values: ['on'], vars: Array.from({ length: 4295 }, (_, i) => ({ id: 'x' + i })), + linear: [{ limit: 5000000, terms: Array.from({ length: 4295 }, (_, i) => ['x' + i, 'on', 1000000]) }], flags: { exact_limit: 0 } }; +error(run(weighted)); +const warm = { ...weighted, start: Object.fromEntries(weighted.vars.map(v => [v.id, 'on'])) }; +error(run(warm)); + +// 2048^3 wrapped to zero; reject the expansion before allocating its tuples. +const values = Array.from({ length: 2048 }, (_, i) => 'v' + i); +const table = { probbit_ir: 1, values, vars: [{ id: 'a' }, { id: 'b' }, { id: 'c' }], tables: [{ vars: ['a', 'b', 'c'], allow: [] }] }; +const tr = run(table); error(tr); assert.equal(tr.out.error.code, 'limit'); + +// k rows existed but were empty: reserving k*k f64s before validating them +// trapped on wasm32 even though the malformed document itself was small. +const hugeAlphabet = Array.from({ length: 65535 }, (_, i) => 'v' + i); +const ragged = { probbit_ir: 1, values: hugeAlphabet, vars: [{ id: 'a' }, { id: 'b' }], + pairs: [{ i: 'a', j: 'b', table: hugeAlphabet.map(() => []) }] }; +const rr = run(ragged); error(rr); assert.equal(rr.out.error.path, 'pairs[0].table[0]'); + +function contract(r) { + const j = r.out; + if (!j.plan) { assert.equal(j.released_plan, undefined); return; } + assert.equal(j.plan_status, ['exact', 'diagnostics_passed'].includes(j.verdict) ? 'released' : j.verdict === 'partial' ? 'partial' : 'diagnostic'); + assert.deepEqual(Object.keys(j.released_plan).sort(), [...j.released].sort()); + for (const [id, value] of Object.entries(j.released_plan)) assert.equal(value, j.plan[id]); + if (j.verdict === 'refused') assert.deepEqual(j.released_plan, {}); +} +const transient = new Set(['ms', 'sample_ms', 'gate_ms', 'site_updates_per_s', 'process_cpu_ms', 'peak_rss_mb', 'nice', 'phases', 'threads']); +function stable(j) { + if (Array.isArray(j)) return j.map(stable); + if (j && typeof j === 'object') return Object.fromEntries(Object.entries(j).filter(([k]) => !transient.has(k)).map(([k, v]) => [k, stable(v)])); + return j; +} +let parity = 0; +function compare(op, doc, flags = {}, label = '') { + const w = call(op, JSON.stringify({ ...doc, flags })); contract(w); + if (!native) return w; + const cmd = ['decide', 'run', 'evaluate'][op]; + const args = [cmd, '--threads', '4']; + for (const [key, value] of Object.entries(flags)) args.push('--' + key.replaceAll('_', '-'), String(value)); + const { PROBBIT_CONFIG: _config, PROBBIT_THREADS: _threads, ...env } = process.env; + const r = spawnSync(native, args, { input: JSON.stringify(doc), encoding: 'utf8', env }); + assert.equal(r.status, w.code, `${label}: ${r.stderr}`); + assert.deepEqual(stable(JSON.parse(r.stdout)), stable(w.out), label); + parity++; return w; +} +compare(1, base, {}, 'exact'); +compare(1, base, { op: 'sample', sweeps: 1, polish_ms: 0 }, 'refusal'); +compare(1, precedence, {}, 'infeasible'); +const demo = call(3, '{"tasks":300,"seed":7}').out; +compare(0, demo, { mode: 'sample', sweeps: 1000, polish_sweeps: 50 }, 'router partial'); +for (const [file, op] of [['knapsack-20.json', 1], ['agent-plan-6.json', 1], ['evaluate/support-12.json', 2]]) { + const doc = JSON.parse(readFileSync(new URL('../../examples/' + file, import.meta.url))); + compare(op, doc, {}, file + ' exact'); + compare(op, doc, { op: 'sample', sweeps: 1000, polish_sweeps: 50 }, file + ' fixed-work'); +} +if (native) { + const n = spawnSync(native, ['demo', '--tasks', '1', '--seed', '4294967296'], { encoding: 'utf8' }); + assert.equal(n.status, 0); assert.deepEqual(JSON.parse(n.stdout), seed64.out); parity++; +} +console.log(JSON.stringify({ wasm_calls: calls, native_parity_cases: parity, boundaries: 'passed' })); diff --git a/python/probbit.py b/python/probbit.py index b90dcb0..b744af1 100644 --- a/python/probbit.py +++ b/python/probbit.py @@ -22,17 +22,19 @@ ProbbitInputError exit 2: the input or a flag was rejected (.code / .path / .message from the structured error object; flag errors carry the CLI's stderr line in .message and code "flag") ProbbitNumericError exit 3 with an {"error": {"code": "numeric"}} object: a computed quantity was not finite (no answer) + ProbbitControlError live exit 4: locked / paused / retired (.code / .path / .message); no event appended ProbbitTimeout the call passed `timeout_s` (the process was killed) ProbbitJudgeError `evaluate`'s judge failed (HTTP error, unreachable, not JSON, a missing key variable): no answer ProbbitError anything else (missing binary, a crash, unparseable output) Keyword flags map to CLI flags: budget_ms=200 -> --budget-ms 200; collective=False -> --collective off (collective, cluster, cycles take on / off); any other boolean is a switch: summary=True -> --summary, pretty=True -> --pretty, False leaves it out. The binary: `binary=` argument, else $PROBBIT_BIN, else `probbit` on PATH, else ../target/release/probbit next to this file. +An explicitly selected binary that is missing or not executable raises; it never falls back to another version. """ import hashlib, json, os, shutil, subprocess, tempfile, urllib.error, urllib.parse, urllib.request __all__ = ["run", "exact", "sample", "decide", "demo", "evaluate", "persona_init", "persona_turn", "persona_replay", "persona_fuzz", "persona_prove", "live_event", "live_verify", "find_binary", "ProbbitError", - "ProbbitInputError", "ProbbitNumericError", "ProbbitTimeout", "ProbbitJudgeError"] + "ProbbitInputError", "ProbbitNumericError", "ProbbitTimeout", "ProbbitJudgeError", "ProbbitControlError"] class ProbbitError(Exception): @@ -48,6 +50,12 @@ def __str__(self): return f"{self.code} at {self.path or ''}: {self.message}" +class ProbbitControlError(ProbbitError): + """A live strand is locked, paused or retired (exit 4). Nothing was appended.""" + def __init__(self, code, path, message, exit_code=4, stdout="", stderr=""): + super().__init__(message, exit_code, stdout, stderr); self.code, self.path = code, path + + class ProbbitNumericError(ProbbitError): def __init__(self, path, message, exit_code=3, stdout="", stderr=""): super().__init__(message, exit_code, stdout, stderr); self.path = path @@ -63,13 +71,30 @@ def __init__(self, message, status=None, body=""): def find_binary(binary=None): - for b in (binary, os.environ.get("PROBBIT_BIN"), shutil.which("probbit"), + # An explicit selection is authoritative: never silently run another installed binary. + selected = binary if binary is not None else os.environ.get("PROBBIT_BIN") + if selected is not None: + path = os.fspath(selected) + found = path if os.path.isfile(path) and os.access(path, os.X_OK) else shutil.which(path) + if found: + return os.path.abspath(found) + raise ProbbitError(f"selected probbit binary is missing or not executable: {path}") + for b in (shutil.which("probbit"), os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "target", "release", "probbit")): if b and os.path.isfile(b) and os.access(b, os.X_OK): return b raise ProbbitError("probbit binary not found: pass binary=, set PROBBIT_BIN, or build with `cargo build --release`") +def _execute(args, *, timeout_s=None, text=None): + try: + return subprocess.run(args, input=text, capture_output=True, encoding="utf-8", timeout=timeout_s) + except subprocess.TimeoutExpired as e: + raise ProbbitTimeout(f"probbit call passed timeout_s={timeout_s}") from e + except OSError as e: + raise ProbbitError(f"cannot execute selected probbit binary: {e}") from e + + _ON_OFF = ("collective", "cluster", "cycles") @@ -92,10 +117,7 @@ def _flags(flags): def _call(cmd, doc, flags, timeout_s, binary): text = doc if isinstance(doc, str) else json.dumps(doc) args = [find_binary(binary), *cmd, *_flags(flags)] - try: - p = subprocess.run(args, input=text, capture_output=True, encoding="utf-8", timeout=timeout_s) # probbit reads and writes UTF-8, whatever the locale - except subprocess.TimeoutExpired as e: - raise ProbbitTimeout(f"probbit {' '.join(cmd)} passed timeout_s={timeout_s}") from e + p = _execute(args, text=text, timeout_s=timeout_s) try: out = json.loads(p.stdout) if p.stdout.strip() else None except ValueError: @@ -140,7 +162,7 @@ def decide(problem, *, timeout_s=None, binary=None, **flags): def demo(tasks=24, seed=1, hard=False, binary=None): """`probbit demo`: a synthetic agent-routing document (dict).""" args = [find_binary(binary), "demo", "--tasks", str(tasks), "--seed", str(seed)] + (["--hard"] if hard else []) - p = subprocess.run(args, capture_output=True, text=True) + p = _execute(args) if p.returncode != 0: raise ProbbitError(f"probbit demo exited {p.returncode}: {p.stderr.strip()}", p.returncode, p.stdout, p.stderr) return json.loads(p.stdout) @@ -243,10 +265,7 @@ def _json_file(d, name, doc): def _persona_call(args, timeout_s, binary, lines=False, answers=(0,)): - try: - p = subprocess.run([find_binary(binary), "persona", *args], capture_output=True, encoding="utf-8", timeout=timeout_s) - except subprocess.TimeoutExpired as e: - raise ProbbitTimeout(f"probbit persona {args[0]} passed timeout_s={timeout_s}") from e + p = _execute([find_binary(binary), "persona", *args], timeout_s=timeout_s) try: out = [json.loads(x) for x in p.stdout.splitlines()] if lines and p.returncode == 0 else (json.loads(p.stdout) if p.stdout.strip() else None) except ValueError: @@ -276,7 +295,7 @@ def persona_turn(persona, state, inputs=None, *, timeout_s=None, binary=None, ** ProbbitInputError (code "persona", .path, .message); a refused or fallback stance is an answer.""" with tempfile.TemporaryDirectory() as d: nxt = os.path.join(d, "next.json") - args = ["turn", _persona_path(persona, d), "--state", _json_file(d, "state.json", state), "--inputs", _json_file(d, "inputs.json", inputs or {}), + args = ["turn", _persona_path(persona, d), "--state", _json_file(d, "state.json", state), "--inputs", _json_file(d, "inputs.json", {} if inputs is None else inputs), "--out", nxt, *_flags(flags)] stance = _persona_call(args, timeout_s, binary) with open(nxt, encoding="utf-8") as f: @@ -314,7 +333,9 @@ def persona_fuzz(persona, never=None, props=None, seeds="0-99", *, timeout_s=Non a character property -> the probbit_persona_fuzz document (a dict; "found" says whether any rule broke, each property its shortest counterexample with the replay command). never: one rule in habit syntax ({"when": {...}, "then": {...}}, or one line of YAML); props: a list of rules (each with an optional "id") or a props file. seeds: "0-99", "1,4,9" or a list. Flags: - fuzz_seed, scripts, depth, beam, grid and hours (lists or comma strings), threads. A counterexample is an answer, not an error; + fuzz_seed, scripts, depth, beam, grid and hours (lists or comma strings), threads, fixture_src. + fixture_src="env:synthetic" (or "human:synthetic") explicitly authorizes synthetic test fixtures under reward_from; it never + relabels production evidence. Without it source-required counterexamples carry null replay/explain commands. A counterexample is an answer, not an error; it tests the stance, not the words a model writes.""" with tempfile.TemporaryDirectory() as d: return _persona_call(["fuzz", _persona_path(persona, d), *_rules(d, never, props, seeds, flags)], timeout_s, binary, answers=(0, 1)) @@ -330,10 +351,7 @@ def persona_prove(persona, never=None, props=None, seeds="0-99", *, timeout_s=No # ---------------------------------------------------------------- live: a resident individual (docs/persona.md §5.7) def _live_call(args, timeout_s, binary, answers=(0,)): - try: - p = subprocess.run([find_binary(binary), "live", *args], capture_output=True, encoding="utf-8", timeout=timeout_s) - except subprocess.TimeoutExpired as e: - raise ProbbitTimeout(f"probbit live passed timeout_s={timeout_s}") from e + p = _execute([find_binary(binary), "live", *args], timeout_s=timeout_s) try: out = json.loads(p.stdout) if p.stdout.strip() else None except ValueError: @@ -343,6 +361,8 @@ def _live_call(args, timeout_s, binary, answers=(0,)): if err: raise ProbbitInputError(err.get("code"), err.get("path"), err.get("message"), 2, p.stdout, p.stderr) raise ProbbitInputError("flag", None, p.stderr.strip().removeprefix("probbit: "), 2, p.stdout, p.stderr) + if p.returncode == 4 and isinstance(err, dict): + raise ProbbitControlError(err.get("code"), err.get("path"), err.get("message"), 4, p.stdout, p.stderr) if p.returncode in answers and isinstance(out, dict) and not err: return out raise ProbbitError(f"probbit live exited {p.returncode}: {p.stderr.strip()[:300]}", p.returncode, p.stdout, p.stderr) @@ -363,15 +383,25 @@ def live_event(persona, state=None, event=None, *, seed=None, strand=None, timeo state = _persona_call(["init", path] + (["--seed", str(seed)] if seed is not None else []), timeout_s, binary) sf, ev = _json_file(d, "state.json", state), os.path.join(d, "event.jsonl") with open(ev, "w", encoding="utf-8") as f: - f.write(json.dumps(event or {}, ensure_ascii=False) + "\n") + f.write(json.dumps({} if event is None else event, ensure_ascii=False) + "\n") args = [path, "--state", sf, "--clock", "fixed", "--events", ev] + (["--strand", os.fspath(strand)] if strand is not None else []) stance = _live_call(args, timeout_s, binary) with open(sf, encoding="utf-8") as f: out = {"stance": stance, "state": json.load(f)} if strand is not None: with open(strand, encoding="utf-8") as f: - last = f.read().splitlines()[-1] - out["strand"] = {"path": os.fspath(strand), "events": json.loads(last).get("n", 0), "head": "sha256:" + hashlib.sha256(last.encode("utf-8")).hexdigest()} + # A later writer may be appending while this receipt is read. Only complete + # lines are candidates; our own committed event always has its newline. + matches = [(line.rstrip("\n"), json.loads(line)) for line in f if line.endswith("\n") and line.strip()] + committed = [(line, row) for line, row in matches if row.get("state") == out["state"]["digest"] and "n" in row] + if not committed: + raise ProbbitError("the committed event was not found in the strand") + line, row = committed[-1] + # The CLI adds a checkpoint after each 1000th event; that checkpoint is our head. + index = next(i for i in range(len(matches) - 1, -1, -1) if matches[i][0] == line) + if index + 1 < len(matches) and matches[index + 1][1].get("checkpoint") == row["n"]: + line = matches[index + 1][0] + out["strand"] = {"path": os.fspath(strand), "events": row["n"], "head": "sha256:" + hashlib.sha256(line.encode("utf-8")).hexdigest()} return out diff --git a/python/test_agent_harness.py b/python/test_agent_harness.py new file mode 100644 index 0000000..c25090a --- /dev/null +++ b/python/test_agent_harness.py @@ -0,0 +1,20 @@ +"""Keep the public host-enforcement example in the Python/CI regression suite.""" +import os +from pathlib import Path +import subprocess +import sys +import unittest +import probbit + + +class AgentHarness(unittest.TestCase): + def test_runnable_host_example(self): + root = Path(__file__).resolve().parent.parent + environment = dict(os.environ, PROBBIT_BIN=probbit.find_binary()) + result = subprocess.run([sys.executable, "-m", "unittest", "discover", "-s", "examples/agent-harness", "-v"], + cwd=root, env=environment, capture_output=True, encoding="utf-8", timeout=30) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + + +if __name__ == "__main__": + unittest.main() diff --git a/python/test_mcp.py b/python/test_mcp.py index 74fcbc6..9f36930 100644 --- a/python/test_mcp.py +++ b/python/test_mcp.py @@ -71,6 +71,40 @@ def call(self, name, arguments, meta=None): self.assertIn("result", r, r) return r["result"] + def test_invalid_envelope_params_do_not_execute_and_server_recovers(self): + for msg, code in [ + ({"jsonrpc": "2.0", "id": 5, "method": "ping", "params": []}, -32602), + ({"jsonrpc": "2.0", "id": 5, "method": "ping", "params": {"_meta": False}}, -32602), + ({"jsonrpc": "2.0", "id": 0.5, "method": "ping"}, -32600), + ({"jsonrpc": "2.0", "id": 5, "method": "tools/list", "params": {"_meta": dict(MODERN, **{"io.modelcontextprotocol/clientCapabilities": False})}}, -32602), + ]: + self.c.send_raw(json.dumps(msg)) + self.assertEqual(self.c.recv()["error"]["code"], code) + self.assertIn("result", self.c.request("ping")) + + def test_fuzz_explicit_synthetic_source_replays_through_strict_persona_tool(self): + self.legacy() + persona = {"probbit_persona": 1, "identity": {"name": "Fixture", "version": "1"}, + "traits": [{"id": "action", "levels": ["retry", "ask"], "logw": [1, 0]}], + "inputs": [{"id": "praise", "kind": "flag"}, {"id": "criticism", "kind": "flag"}], + "learning": {"from": ["praise", "criticism"], "traits": ["action"], "rate": 3, "step_cap": 2, "total_cap": 2}, + "reward_from": ["env"]} + args = {"persona": persona, "never": {"then": {"action": ["retry"]}}, "seeds": [0], "scripts": 0, "depth": 3, "threads": 1} + raw = self.call("probbit_persona_fuzz", args)["structuredContent"] + self.assertIsNone(raw["properties"][0]["shortest"]["replay"]) + f = self.call("probbit_persona_fuzz", dict(args, fixture_src="env:synthetic")) + self.assertFalse(f["isError"]) + shortest = f["structuredContent"]["properties"][0]["shortest"] + state = self.call("probbit_persona_init", {"persona": persona, "seed": 0})["structuredContent"] + for event in shortest["script"]: + turn = self.call("probbit_persona_turn", {"persona": persona, "state": state, "inputs": event}) + self.assertFalse(turn["isError"]) + state = turn["structuredContent"]["state"] + self.assertEqual(turn["structuredContent"]["stance"], shortest["stance"]) + self.assertTrue(self.call("probbit_persona_fuzz", dict(args, fixture_src="env:production"))["isError"]) + bad = self.call("probbit_persona_turn", {"persona": persona, "state": state, "inputs": {"praise": True, "src": "self"}}) + self.assertTrue(bad["isError"]) + def test_legacy_handshake_lists_eight_tools(self): init = self.legacy() self.assertEqual(init["protocolVersion"], "2025-06-18"); self.assertEqual(init["serverInfo"]["name"], "probbit") diff --git a/python/test_persona.py b/python/test_persona.py index c56cf33..ae659dc 100644 --- a/python/test_persona.py +++ b/python/test_persona.py @@ -140,5 +140,75 @@ def test_live_event_logs_a_strand_that_verifies(self): self.assertEqual(e.exception.path, "events[0].event.elapsed_hours") +class BoundaryErrors(unittest.TestCase): + def test_explicit_missing_binary_does_not_fall_back(self): + with self.assertRaises(probbit.ProbbitError): + probbit.persona_init(load("tutor.json"), binary="/not/a/probbit/binary") + + def test_falsey_malformed_events_are_not_replaced_with_empty_objects(self): + doc = load("tutor.json") + state = probbit.persona_init(doc) + for event in (False, [], "", 0): + with self.subTest(event=event): + with self.assertRaises(probbit.ProbbitInputError): + probbit.persona_turn(doc, state, event) + with self.assertRaises(probbit.ProbbitInputError): + probbit.live_event(doc, state, event) + + def test_replay_error_after_a_good_turn_retains_its_type(self): + with self.assertRaises(probbit.ProbbitInputError) as err: + probbit.persona_replay(load("tutor.json"), [{}, {"sentiment": "not-a-level"}]) + self.assertEqual(err.exception.code, "persona") + self.assertIn("sentiment", err.exception.path) + + def test_checkpoint_event_receipt_names_the_committed_event_and_checkpoint(self): + import pathlib, subprocess, tempfile + doc = {"probbit_persona": 1, "identity": {"name": "Receipt", "version": "1"}, + "traits": [{"id": "action", "levels": ["retry", "ask"], "logw": [1, 0]}]} + with tempfile.TemporaryDirectory() as directory: + root = pathlib.Path(directory) + policy, state, strand, events = [root / name for name in ("policy.json", "state.json", "receipt.strand", "events.jsonl")] + policy.write_text(json.dumps(doc), encoding="utf-8") + state.write_text(json.dumps(probbit.persona_init(doc)), encoding="utf-8") + events.write_text("{}\n" * 999, encoding="utf-8") + subprocess.run([probbit.find_binary(), "live", str(policy), "--state", str(state), "--clock", "fixed", "--events", str(events), "--strand", str(strand)], check=True, capture_output=True) + receipt = probbit.live_event(doc, json.loads(state.read_text()), {}, strand=strand) + verified = probbit.live_verify(strand) + self.assertEqual(receipt["strand"]["events"], 1000) + self.assertEqual(receipt["strand"]["head"], verified["last_line"]) + self.assertEqual(verified["checkpoints"], 1) + + def test_later_control_or_partial_append_cannot_replace_our_event_receipt(self): + import hashlib, pathlib, subprocess, tempfile + from unittest.mock import patch + original = probbit._live_call + with tempfile.TemporaryDirectory() as directory: + strand = pathlib.Path(directory) / "receipt.strand" + def after_event(*args, **kwargs): + result = original(*args, **kwargs) + subprocess.run([probbit.find_binary(), "live", "control", str(strand), "pause", "--by", "human:owner", "--reason", "test"], check=True, capture_output=True) + with strand.open("a", encoding="utf-8") as stream: + stream.write('{"incomplete":') # a subsequent writer has not finished yet + return result + with patch.object(probbit, "_live_call", side_effect=after_event): + receipt = probbit.live_event(load("tutor.json"), event={}, strand=strand) + event_line = strand.read_text().splitlines()[1] + expected_head = "sha256:" + hashlib.sha256(event_line.encode()).hexdigest() + self.assertEqual(receipt["strand"], {"path": str(strand), "events": 1, "head": expected_head}) + + def test_pause_is_typed_and_never_appends(self): + import pathlib, subprocess, tempfile + with tempfile.TemporaryDirectory() as directory: + strand = pathlib.Path(directory) / "test.strand" + doc = load("tutor.json") + first = probbit.live_event(doc, event={}, strand=strand) + subprocess.run([probbit.find_binary(), "live", "control", str(strand), "pause", "--by", "human:owner", "--reason", "test"], check=True, capture_output=True) + before = strand.read_bytes() + with self.assertRaises(probbit.ProbbitControlError) as err: + probbit.live_event(doc, first["state"], {}, strand=strand) + self.assertEqual(err.exception.code, "paused") + self.assertEqual(strand.read_bytes(), before) + + if __name__ == "__main__": unittest.main() diff --git a/scripts/bench/test_install_ps1.ps1 b/scripts/bench/test_install_ps1.ps1 index 604967f..7466cfc 100644 --- a/scripts/bench/test_install_ps1.ps1 +++ b/scripts/bench/test_install_ps1.ps1 @@ -3,12 +3,11 @@ # installs it with -DownloadBase, then the `irm | iex` form (environment variables), then the tampered copy (must fail). # pwsh -File scripts/bench/test_install_ps1.ps1 -Srv [-Tag v0.2.0] [-AddToPath] # powershell -ExecutionPolicy Bypass -File scripts/bench/test_install_ps1.ps1 -Srv -param([Parameter(Mandatory = $true)][string]$Srv, [string]$Tag = 'v0.2.0', [switch]$AddToPath) +param([Parameter(Mandatory = $true)][string]$Srv, [string]$Tag = 'v0.8.0', [switch]$AddToPath) $ErrorActionPreference = 'Stop' "PowerShell $($PSVersionTable.PSVersion) ($($PSVersionTable.PSEdition))" -# Windows PowerShell 5.1 adds a UTF-8 byte-order mark to text it pipes into a native program, which probbit 0.2.0 rejects -# (docs/agents.md, PowerShell): there the demo pipeline below goes through cmd.exe. -$desktop = $PSVersionTable.PSEdition -ne 'Core' +$oldPath = $env:Path +$oldUserPath = [Environment]::GetEnvironmentVariable('Path', 'User') $l = [System.Net.Sockets.TcpListener]::new([System.Net.IPAddress]::Loopback, 0); $l.Start(); $port = $l.LocalEndpoint.Port; $l.Stop() $server = Start-Process -FilePath python -ArgumentList @('scripts/bench/serve.py', $Srv, "$port") -PassThru -WindowStyle Hidden $base = "http://127.0.0.1:$port" @@ -24,8 +23,8 @@ try { "`$ (Get-Command probbit).Source: $($cmd.Source)" if ((Resolve-Path $cmd.Source).Path -ne (Resolve-Path (Join-Path $d1 'probbit.exe')).Path) { throw "FAIL: probbit resolves to $($cmd.Source)" } "`$ probbit version: $(probbit version)" - if ($desktop) { $how = 'cmd /c "probbit demo --tasks 12 | probbit decide"'; $d = cmd /c "probbit demo --tasks 12 | probbit decide" | ConvertFrom-Json } - else { $how = 'probbit demo --tasks 12 | probbit decide'; $d = probbit demo --tasks 12 | probbit decide | ConvertFrom-Json } + $how = 'probbit demo --tasks 12 | probbit decide' + $d = probbit demo --tasks 12 | probbit decide | ConvertFrom-Json "`$ $how | ConvertFrom-Json: exit $LASTEXITCODE, verdict $($d.verdict), released $(@($d.released).Count) of $($d.tasks)" if ($LASTEXITCODE -ne 0 -or $d.verdict -ne 'exact') { throw 'FAIL: demo pipeline' } if ($AddToPath) { @@ -53,7 +52,21 @@ try { $failed = $false try { & .\install.ps1 -DownloadBase "$base/good" -InstallDir (Join-Path $root 'bin4') } catch { $failed = $true; "rejected: $($_.Exception.Message)" } if (-not $failed) { throw 'FAIL: accepted' } + + '== 5. failed update preserves an existing install' + $before = (Get-FileHash -LiteralPath (Join-Path $d1 'probbit.exe') -Algorithm SHA256).Hash + $failed = $false + try { & .\install.ps1 -DownloadBase "$base/bad" -Version $Tag -InstallDir $d1 } catch { $failed = $true } + if (-not $failed) { throw 'FAIL: accepted damaged update' } + if ((Get-FileHash -LiteralPath (Join-Path $d1 'probbit.exe') -Algorithm SHA256).Hash -ne $before) { throw 'FAIL: existing install changed' } + + '== 6. valid replacement uses the same destination' + & .\install.ps1 -DownloadBase "$base/good" -Version $Tag -InstallDir $d1 + if ((Get-FileHash -LiteralPath (Join-Path $d1 'probbit.exe') -Algorithm SHA256).Hash -ne $before) { throw 'FAIL: replacement changed the binary' } "install.ps1: all checks passed (PowerShell $($PSVersionTable.PSVersion))" } finally { Stop-Process -Id $server.Id -Force -ErrorAction SilentlyContinue + $env:Path = $oldPath + if ($AddToPath) { [Environment]::SetEnvironmentVariable('Path', $oldUserPath, 'User') } + if ($root -and (Test-Path -LiteralPath $root)) { Remove-Item -LiteralPath $root -Recurse -Force } } diff --git a/scripts/bench/test_install_sh.sh b/scripts/bench/test_install_sh.sh index 837c4fa..595b086 100755 --- a/scripts/bench/test_install_sh.sh +++ b/scripts/bench/test_install_sh.sh @@ -4,7 +4,7 @@ # .sha256 is wrong (must fail and install nothing). # scripts/bench/test_install_sh.sh [tag] SH=dash scripts/bench/... to pick the shell set -eu -BIN=$1 TARGET=$2 TAG=${3:-v0.2.0} +BIN=$1 TARGET=$2 TAG=${3:-v$(node -p "require('./npm/package.json').version")} SH=${SH:-sh} PY=${PY:-$(command -v python3 || command -v python)} WORK=$(mktemp -d 2> /dev/null || mktemp -d -t probbitinst) @@ -34,7 +34,7 @@ echo "\$ probbit demo --tasks 12 | probbit decide: exit $?, $("$PY" -c 'import j echo "== 2. default directory, scratch HOME" HOME="$WORK/home" PROBBIT_DOWNLOAD_BASE="$BASE/good" PROBBIT_VERSION="$TAG" "$SH" install.sh -if [ -d /usr/local/bin ] && [ -w /usr/local/bin ]; then want=/usr/local/bin/probbit; else want="$WORK/home/.local/bin/probbit"; fi +want="$WORK/home/.local/bin/probbit" [ -x "$want" ] || { echo "FAIL: expected $want"; exit 1; } echo "ok: installed to $want ($("$want" version))" @@ -48,4 +48,22 @@ echo "ok: rejected, $WORK/badbin/probbit absent" echo "== 4. PROBBIT_DOWNLOAD_BASE without PROBBIT_VERSION must fail" if PROBBIT_DOWNLOAD_BASE="$BASE/good" PROBBIT_INSTALL_DIR="$WORK/bin4" "$SH" install.sh; then echo "FAIL: accepted"; exit 1; fi echo "ok: rejected" + +echo "== 5. failed update preserves an existing install" +before=$("$PY" -c 'import hashlib,sys; print(hashlib.sha256(open(sys.argv[1],"rb").read()).hexdigest())' "$WORK/bin/probbit") +if PROBBIT_DOWNLOAD_BASE="$BASE/bad" PROBBIT_VERSION="$TAG" PROBBIT_INSTALL_DIR="$WORK/bin" "$SH" install.sh; then echo "FAIL: accepted damaged update"; exit 1; fi +after=$("$PY" -c 'import hashlib,sys; print(hashlib.sha256(open(sys.argv[1],"rb").read()).hexdigest())' "$WORK/bin/probbit") +[ "$before" = "$after" ] || { echo "FAIL: existing install changed"; exit 1; } + +echo "== 6. invalid release path is rejected before download" +if PROBBIT_DOWNLOAD_BASE="$BASE/good" PROBBIT_VERSION='../bad' PROBBIT_INSTALL_DIR="$WORK/bin" "$SH" install.sh; then echo "FAIL: invalid version accepted"; exit 1; fi + +echo "== 7. checksummed but unrunnable candidate must preserve the working binary" +mkdir -p "$WORK/unrunnable" +printf '#!/bin/sh\nexit 25\n' > "$WORK/unrunnable/probbit" +chmod +x "$WORK/unrunnable/probbit" +sh scripts/bench/package_like_release.sh "$WORK/unrunnable/probbit" "$TARGET" "$TAG" "$WORK/srv/unrunnable/$TAG" > /dev/null +if PROBBIT_DOWNLOAD_BASE="$BASE/unrunnable" PROBBIT_VERSION="$TAG" PROBBIT_INSTALL_DIR="$WORK/bin" "$SH" install.sh; then echo "FAIL: unrunnable candidate accepted"; exit 1; fi +after=$("$PY" -c 'import hashlib,sys; print(hashlib.sha256(open(sys.argv[1],"rb").read()).hexdigest())' "$WORK/bin/probbit") +[ "$before" = "$after" ] || { echo "FAIL: unrunnable candidate replaced working binary"; exit 1; } echo "install.sh: all checks passed ($SH, $TARGET)" diff --git a/scripts/bench/test_npm.sh b/scripts/bench/test_npm.sh index 38f2930..79b4061 100755 --- a/scripts/bench/test_npm.sh +++ b/scripts/bench/test_npm.sh @@ -5,7 +5,7 @@ # Then an --ignore-scripts install (fetched on first run), PROBBIT_BINARY, and a tampered .sha256 (the install must fail). # scripts/bench/test_npm.sh [tag] set -eu -BIN=$1 TARGET=$2 TAG=${3:-v0.2.0} +BIN=$1 TARGET=$2 TAG=${3:-v$(node -p "require('./npm/package.json').version")} PY=${PY:-$(command -v python3 || command -v python)} WORK=$(mktemp -d 2> /dev/null || mktemp -d -t probbitnpm) SRV="" @@ -28,6 +28,20 @@ done BASE="http://127.0.0.1:$PORT" case "$(uname -s)" in MINGW* | MSYS* | CYGWIN*) gbin() { echo "$1"; } ;; *) gbin() { echo "$1/bin"; } ;; esac NPM_FLAGS="--no-audit --no-fund --foreground-scripts" +NPM_MAJOR=$(npm --version | cut -d. -f1) +# npm 12 identifies a local tarball by its resolved path, not its self-reported package name. +# Allow only this exact test artifact; a registry-style "probbit" opt-in must not match a local tarball. +install_package() { + if [ "$NPM_MAJOR" -ge 12 ]; then npm install -g $NPM_FLAGS "--allow-scripts=$TGZ" "$@" + else npm install -g $NPM_FLAGS "$@"; fi +} +# Keep every fallback local, and prove the candidate bytes rather than accidentally testing a published release. +export PROBBIT_DOWNLOAD_BASE="$BASE/good" PROBBIT_TARGET="$TARGET" PROBBIT_VERSION="$TAG" +assert_candidate() { + ROOT=$(npm root -g --prefix "$1") + case "$TARGET" in *windows*) EXE=probbit.exe ;; *) EXE=probbit ;; esac + node -e 'const fs=require("fs"); if (!fs.readFileSync(process.argv[1]).equals(fs.readFileSync(process.argv[2]))) throw Error("installed binary differs from candidate");' "$ROOT/probbit/vendor/$EXE" "$BIN_ABS" +} echo "node $(node --version), npm $(npm --version), target $TARGET" (cd npm && npm pack --pack-destination "$WORK" > /dev/null) @@ -37,7 +51,8 @@ tar tzf "$TGZ" | sed 's/^/ /' echo "== 1. npm install -g $(basename "$TGZ") (PROBBIT_DOWNLOAD_BASE=$BASE/good)" P1="$WORK/prefix1" -PROBBIT_DOWNLOAD_BASE="$BASE/good" npm install -g --prefix "$P1" $NPM_FLAGS "$TGZ" +install_package --prefix "$P1" "$TGZ" +assert_candidate "$P1" OLDPATH=$PATH PATH="$(gbin "$P1"):$OLDPATH" echo "\$ which probbit: $(command -v probbit)" @@ -58,17 +73,24 @@ echo "ok: uninstalled, $(gbin "$P1")/probbit absent" echo "== 2. npm install -g --ignore-scripts: the binary is fetched on first run" P2="$WORK/prefix2" -npm install -g --prefix "$P2" $NPM_FLAGS --ignore-scripts "$TGZ" +install_package --prefix "$P2" --ignore-scripts "$TGZ" PROBBIT_DOWNLOAD_BASE="$BASE/good" "$(gbin "$P2")/probbit" version +assert_candidate "$P2" "$(gbin "$P2")/probbit" demo --tasks 12 | "$(gbin "$P2")/probbit" decide > /dev/null echo "ok: first run fetched it; second run used it" echo "== 3. PROBBIT_BINARY= (no download)" P3="$WORK/prefix3" -PROBBIT_BINARY="$BIN_ABS" PROBBIT_DOWNLOAD_BASE="http://127.0.0.1:9/nowhere" npm install -g --prefix "$P3" $NPM_FLAGS "$TGZ" +(export PROBBIT_BINARY="$BIN_ABS" PROBBIT_DOWNLOAD_BASE="http://127.0.0.1:9/nowhere"; install_package --prefix "$P3" "$TGZ") +assert_candidate "$P3" "$(gbin "$P3")/probbit" version echo "== 4. tampered .sha256: npm install must fail" -if PROBBIT_DOWNLOAD_BASE="$BASE/bad" npm install -g --prefix "$WORK/prefix4" $NPM_FLAGS "$TGZ"; then fail "a wrong checksum was accepted"; fi +if (export PROBBIT_DOWNLOAD_BASE="$BASE/bad"; install_package --prefix "$WORK/prefix4" "$TGZ"); then fail "a wrong checksum was accepted"; fi echo "ok: rejected" + +echo "== 5. --ignore-scripts with bad checksum must fail on first use" +P5="$WORK/prefix5" +install_package --prefix "$P5" --ignore-scripts "$TGZ" +if PROBBIT_DOWNLOAD_BASE="$BASE/bad" "$(gbin "$P5")/probbit" version; then fail "first run accepted a wrong checksum"; fi echo "npm wrapper: all checks passed ($TARGET)" diff --git a/scripts/tests/test_install_contract.py b/scripts/tests/test_install_contract.py new file mode 100644 index 0000000..45ee2f4 --- /dev/null +++ b/scripts/tests/test_install_contract.py @@ -0,0 +1,64 @@ +"""Network-free installer routing checks. Run with Python 3.9+ on macOS/Linux.""" +import os +from pathlib import Path +import subprocess +import tempfile +import unittest + +ROOT = Path(__file__).resolve().parents[2] + + +class InstallRouting(unittest.TestCase): + def run_fixture(self, os_name, arch, libc, **extra): + with tempfile.TemporaryDirectory(prefix="probbit-target-") as folder: + root = Path(folder) + scripts = { + "uname": '#!/bin/sh\ncase "$1" in -s) echo "$FIXTURE_OS";; -m) echo "$FIXTURE_ARCH";; esac\n', + "ldd": '#!/bin/sh\necho "$FIXTURE_LIBC"\n', + "sysctl": '#!/bin/sh\necho 0\n', + "curl": '#!/bin/sh\necho "FIXTURE_FETCH $*" >&2\nexit 42\n', + } + for name, body in scripts.items(): + script = root / name + script.write_text(body) + script.chmod(0o755) + env = dict(os.environ, PATH=str(root) + os.pathsep + os.environ["PATH"], + FIXTURE_OS=os_name, FIXTURE_ARCH=arch, FIXTURE_LIBC=libc, + PROBBIT_VERSION="v0.8.0", PROBBIT_INSTALL_DIR=str(root / "bin")) + for key in ["PROBBIT_TARGET", "PROBBIT_DOWNLOAD_BASE"]: + env.pop(key, None) + env.update(extra) + result = subprocess.run(["sh", str(ROOT / "install.sh")], env=env, text=True, + stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=10) + self.assertNotEqual(result.returncode, 0) # the fake transport always declines + self.assertFalse((root / "bin" / "probbit").exists()) + return result.stdout + result.stderr + + def test_supported_target_selection(self): + for os_name, arch, libc, target in [ + ("Darwin", "arm64", "", "aarch64-apple-darwin"), + ("Darwin", "x86_64", "", "x86_64-apple-darwin"), + ("Linux", "x86_64", "glibc 2.35", "x86_64-unknown-linux-gnu"), + ("Linux", "x86_64", "musl libc", "x86_64-unknown-linux-musl"), + ("Linux", "aarch64", "glibc 2.35", "aarch64-unknown-linux-musl"), + ]: + with self.subTest(target=target): + output = self.run_fixture(os_name, arch, libc) + self.assertIn("FIXTURE_FETCH", output) + self.assertIn("probbit-v0.8.0-" + target + ".tar.gz", output) + + def test_unsupported_architecture_never_fetches(self): + output = self.run_fixture("Linux", "riscv64", "glibc") + self.assertIn("no prebuilt", output) + self.assertNotIn("FIXTURE_FETCH", output) + + def test_invalid_release_and_target_never_fetch(self): + for values in [{"PROBBIT_VERSION": "../bad"}, {"PROBBIT_TARGET": "../../bad"}]: + with self.subTest(values=values): + output = self.run_fixture("Darwin", "arm64", "", **values) + self.assertIn("invalid", output) + self.assertNotIn("FIXTURE_FETCH", output) + + +if __name__ == "__main__": + unittest.main()