From 2a0fb46b8ba750efcf8b873b9923e548882d4ec5 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Tue, 8 Sep 2026 13:03:44 +0900 Subject: [PATCH 001/108] docs(plan): record 0.2.24 release and deployment evidence (wp5) --- .../evidence/local-installed-verified.json | 1 + .../evidence/pr88-checks.txt | 11 +++ .../evidence/pr89-checks.txt | 21 ++++++ .../evidence/wp5-delivery-state.json | 69 +++++++++++++++++++ .../desktop-c795oh4-baseline-doctor.txt | 8 +++ .../wp5-ssh/desktop-c795oh4-hashes.txt | 4 ++ .../evidence/wp5-ssh/macmini-cf-hashes.txt | 1 + .../evidence/wp5-ssh/suji-hashes.txt | 1 + 8 files changed, 116 insertions(+) create mode 100644 devlog/_plan/260908_narrative_documents/evidence/local-installed-verified.json create mode 100644 devlog/_plan/260908_narrative_documents/evidence/pr88-checks.txt create mode 100644 devlog/_plan/260908_narrative_documents/evidence/pr89-checks.txt create mode 100644 devlog/_plan/260908_narrative_documents/evidence/wp5-delivery-state.json create mode 100644 devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-baseline-doctor.txt create mode 100644 devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-hashes.txt create mode 100644 devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/macmini-cf-hashes.txt create mode 100644 devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/suji-hashes.txt diff --git a/devlog/_plan/260908_narrative_documents/evidence/local-installed-verified.json b/devlog/_plan/260908_narrative_documents/evidence/local-installed-verified.json new file mode 100644 index 00000000..22b62f2c --- /dev/null +++ b/devlog/_plan/260908_narrative_documents/evidence/local-installed-verified.json @@ -0,0 +1 @@ +{"published": 941, "installed": 941, "matched": 941, "missing": [], "mismatched": []} diff --git a/devlog/_plan/260908_narrative_documents/evidence/pr88-checks.txt b/devlog/_plan/260908_narrative_documents/evidence/pr88-checks.txt new file mode 100644 index 00000000..3e2bf846 --- /dev/null +++ b/devlog/_plan/260908_narrative_documents/evidence/pr88-checks.txt @@ -0,0 +1,11 @@ +artifact (macos-latest) pass 17s https://github.com/lidge-jun/codexclaw/actions/runs/34182957486/job/101925566085 +artifact (ubuntu-latest) pass 16s https://github.com/lidge-jun/codexclaw/actions/runs/34182957486/job/101925565925 +artifact (windows-latest) pass 29s https://github.com/lidge-jun/codexclaw/actions/runs/34182957486/job/101925566114 +enforce-target pass 6s https://github.com/lidge-jun/codexclaw/actions/runs/34182957388/job/101925565645 +install (macos-latest) pass 40s https://github.com/lidge-jun/codexclaw/actions/runs/34182957486/job/101925566019 +install (ubuntu-latest) pass 19s https://github.com/lidge-jun/codexclaw/actions/runs/34182957486/job/101925565839 +test (macos-latest, false) pass 2m21s https://github.com/lidge-jun/codexclaw/actions/runs/34182957555/job/101925566526 +test (ubuntu-latest, false) pass 1m30s https://github.com/lidge-jun/codexclaw/actions/runs/34182957555/job/101925566479 +test (windows-latest, false) pass 3m51s https://github.com/lidge-jun/codexclaw/actions/runs/34182957555/job/101925566336 +test (windows-latest, true) pass 3m48s https://github.com/lidge-jun/codexclaw/actions/runs/34182957555/job/101925566623 +wsl pass 10m49s https://github.com/lidge-jun/codexclaw/actions/runs/34182957615/job/101925566627 diff --git a/devlog/_plan/260908_narrative_documents/evidence/pr89-checks.txt b/devlog/_plan/260908_narrative_documents/evidence/pr89-checks.txt new file mode 100644 index 00000000..4e573dd9 --- /dev/null +++ b/devlog/_plan/260908_narrative_documents/evidence/pr89-checks.txt @@ -0,0 +1,21 @@ +artifact (macos-latest) pass 25s https://github.com/lidge-jun/codexclaw/actions/runs/34183660349/job/101927613245 +artifact (macos-latest) pass 15s https://github.com/lidge-jun/codexclaw/actions/runs/34183664416/job/101927624387 +artifact (ubuntu-latest) pass 13s https://github.com/lidge-jun/codexclaw/actions/runs/34183660349/job/101927613288 +artifact (ubuntu-latest) pass 17s https://github.com/lidge-jun/codexclaw/actions/runs/34183664416/job/101927624247 +artifact (windows-latest) pass 28s https://github.com/lidge-jun/codexclaw/actions/runs/34183660349/job/101927613317 +artifact (windows-latest) pass 41s https://github.com/lidge-jun/codexclaw/actions/runs/34183664416/job/101927624101 +enforce-target pass 5s https://github.com/lidge-jun/codexclaw/actions/runs/34183664454/job/101927624745 +install (macos-latest) pass 42s https://github.com/lidge-jun/codexclaw/actions/runs/34183660349/job/101927613277 +install (macos-latest) pass 36s https://github.com/lidge-jun/codexclaw/actions/runs/34183664416/job/101927624195 +install (ubuntu-latest) pass 31s https://github.com/lidge-jun/codexclaw/actions/runs/34183660349/job/101927613061 +install (ubuntu-latest) pass 21s https://github.com/lidge-jun/codexclaw/actions/runs/34183664416/job/101927624294 +test (macos-latest, false) pass 2m56s https://github.com/lidge-jun/codexclaw/actions/runs/34183660308/job/101927613227 +test (macos-latest, false) pass 2m2s https://github.com/lidge-jun/codexclaw/actions/runs/34183664379/job/101927624280 +test (ubuntu-latest, false) pass 1m25s https://github.com/lidge-jun/codexclaw/actions/runs/34183660308/job/101927613219 +test (ubuntu-latest, false) pass 1m24s https://github.com/lidge-jun/codexclaw/actions/runs/34183664379/job/101927624393 +test (windows-latest, false) pass 4m15s https://github.com/lidge-jun/codexclaw/actions/runs/34183660308/job/101927613221 +test (windows-latest, false) pass 4m34s https://github.com/lidge-jun/codexclaw/actions/runs/34183664379/job/101927624130 +test (windows-latest, true) pass 3m38s https://github.com/lidge-jun/codexclaw/actions/runs/34183660308/job/101927613020 +test (windows-latest, true) pass 4m12s https://github.com/lidge-jun/codexclaw/actions/runs/34183664379/job/101927624292 +wsl pass 12m3s https://github.com/lidge-jun/codexclaw/actions/runs/34183660297/job/101927613059 +wsl pass 13m36s https://github.com/lidge-jun/codexclaw/actions/runs/34183664449/job/101927624274 diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp5-delivery-state.json b/devlog/_plan/260908_narrative_documents/evidence/wp5-delivery-state.json new file mode 100644 index 00000000..15ef50c8 --- /dev/null +++ b/devlog/_plan/260908_narrative_documents/evidence/wp5-delivery-state.json @@ -0,0 +1,69 @@ +{ + "repo": "lidge-jun/codexclaw", + "worktree": "/Users/jun/Developer/new/700_projects/codexclaw-narrative", + "version": "0.2.24", + "manifestVersion": "0.2.24+codex.20260908031619", + "prs": { + "narrative": 87, + "pr84Integration": 84, + "releasePrep": 88, + "promotion": 89 + }, + "devHead": "95402628", + "mainHead": "bb8522726c1491b40895cbbdcb91c8e6ec3caba7", + "tag": "v0.2.24", + "mainChecks": { + "ci": 34184492720, + "packed": 34184492791, + "wsl": 34184492713 + }, + "releaseRun": 34185250992, + "releaseURL": "https://github.com/lidge-jun/codexclaw/releases/tag/v0.2.24", + "assets": [ + "candidate-0.2.24.json", + "codexclaw-payload-0.2.24.tar.gz", + "SHA256SUMS" + ], + "sha256sumsVerified": true, + "installs": { + "local": { + "host": "jun mac", + "previous": "0.2.23+codex.20260908004251", + "installed": "0.2.24+codex.20260908031619", + "filesMatched": 941, + "doctor": "PASS", + "hookTrust": 24, + "backup": "/Users/jun/Developer/new/700_projects/codexclaw-narrative/.codexclaw/evidence/narrative-release-0.2.24/rollback/installed-0.2.23" + }, + "macmini-cf": { + "previous": "0.2.22+codex.20260906224615", + "installed": "0.2.24+codex.20260908031619", + "filesMatched": 941, + "doctor": "PASS", + "hookTrust": 24, + "backup": "/Users/junny/codexclaw-deploy-0.2.24-01a07e62/backup" + }, + "suji": { + "previous": "0.2.22+codex.20260906224615", + "installed": "0.2.24+codex.20260908031619", + "filesMatched": 941, + "doctor": "PASS", + "hookTrust": 24, + "backup": "/Users/neuralarcadepro/codexclaw-deploy-0.2.24-01a07e62/backup" + }, + "desktop-c795oh4": { + "previous": "0.2.21+0.2.22 caches", + "installed": "0.2.24+codex.20260908031619", + "filesMatched": 941, + "doctor": "WARN (python store alias; codex features list under non-interactive SSH) — same conditions on the previous 0.2.22 payload; hook-trust 24, install-root PASS", + "hookTrust": 24, + "backup": "C:\\Users\\user\\codexclaw-deploy-0.2.24-01a07e62\\backup", + "note": "first attempt corrupted config.toml via regex; restored from preimage and re-applied with literal replace" + } + }, + "notDeployed": { + "lidge,intmb,cursor": "codex present, no codexclaw install (not targets)", + "oracle,ocx-ci,win,clisu-oracle*": "unreachable/no install in the last probe; state unknown" + }, + "caveat": "Running Codex sessions keep the previously loaded plugin root; new sessions pick up 0.2.24. No hot-reload claim." +} diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-baseline-doctor.txt b/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-baseline-doctor.txt new file mode 100644 index 00000000..3281c411 --- /dev/null +++ b/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-baseline-doctor.txt @@ -0,0 +1,8 @@ +[WARN] ast-grep: python not runnable (the Microsoft Store alias exits 9009) - install Python 3.9+ from python.org +[FAIL] install-root: this payload declares 0.2.22+codex.20260906224615, but the installed root(s) are: C:\Users\user\.codex\plugins\cache\codexclaw\codexclaw\0.2.24+codex.20260908031619. Any session started before the last reinstall is running hooks from a path that no longer exists (STALE-ROOT-01). (repair: codex plugin add @, then RESTART Codex ??a running session keeps the old PLUGIN_ROOT) +[WARN] features: could not read 'codex features list' (repair: ensure the `codex` binary is on PATH, then re-run `cxc doctor`) +overall: FAIL +apply_patch_freeform removed false +apply_patch_streaming_events under development false +apps stable true +apps_mcp_path_override removed false diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-hashes.txt b/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-hashes.txt new file mode 100644 index 00000000..a40c3f59 --- /dev/null +++ b/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-hashes.txt @@ -0,0 +1,4 @@ +ok=941 bad=0 missing=0 +[WARN] ast-grep: python not runnable (the Microsoft Store alias exits 9009) - install Python 3.9+ from python.org +[WARN] features: could not read 'codex features list' (repair: ensure the `codex` binary is on PATH, then re-run `cxc doctor`) +overall: WARN diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/macmini-cf-hashes.txt b/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/macmini-cf-hashes.txt new file mode 100644 index 00000000..8b188d63 --- /dev/null +++ b/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/macmini-cf-hashes.txt @@ -0,0 +1 @@ +ok=941 bad=0 diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/suji-hashes.txt b/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/suji-hashes.txt new file mode 100644 index 00000000..8b188d63 --- /dev/null +++ b/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/suji-hashes.txt @@ -0,0 +1 @@ +ok=941 bad=0 From 2a77559b74bf4448cdb3a841f65420fbae8288e4 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Tue, 8 Sep 2026 13:04:20 +0900 Subject: [PATCH 002/108] docs: close and archive the reader-documents / deep-research / 0.2.24 delivery unit --- .../260908_narrative_documents/000_plan.md | 18 ++++++++++++++++++ .../260908_narrative_documents/001_sources.md | 0 .../002_verifiers.md | 0 .../010_reader_documents.md | 0 .../020_deep_research.md | 0 .../030_pr84_integration.md | 0 .../040_release_0_2_24.md | 0 .../050_deployment.md | 0 .../evidence/check-links.mjs | 0 .../evidence/local-installed-verified.json | 0 .../evidence/pr88-checks.txt | 0 .../evidence/pr89-checks.txt | 0 .../evidence/wp2-audit/verdict.md | 0 .../evidence/wp2-forward/fresh-reader.md | 0 .../evidence/wp2-forward/raw-evidence-dump.md | 0 .../evidence/wp2-forward/self-check.md | 0 .../evidence/wp2-forward/status-report.md | 0 .../evidence/wp3-audit/verdict.md | 0 .../evidence/wp3-forward-run1/gap-matrix.md | 0 .../evidence/wp3-forward-run1/journal.md | 0 .../evidence/wp3-forward-run1/ledger.md | 0 .../evidence/wp3-forward-run1/plan.md | 0 .../wp3-forward-run1/report-source.md | 0 .../evidence/wp3-forward-run1/report.html | 0 .../evidence/wp3-forward-run2/plan.md | 0 .../api-assafelovic__gpt-researcher.json | 0 .../wp3-forward/api-bytedance__deer-flow.json | 0 .../wp3-forward/api-dzhng__deep-research.json | 0 .../api-langchain-ai__deepagents.json | 0 .../api-langchain-ai__open_deep_research.json | 0 .../wp3-forward/api-modelscope__ms-agent.json | 0 .../wp3-forward/api-stanford-oval__storm.json | 0 .../evidence/wp3-forward/gap-matrix.md | 0 .../evidence/wp3-forward/journal.md | 0 .../evidence/wp3-forward/ledger.md | 0 .../evidence/wp3-forward/plan.md | 0 .../evidence/wp3-forward/readme-deer-flow.md | 0 .../wp3-forward/readme-dzhng-deep-research.md | 0 .../wp3-forward/readme-gpt-researcher.md | 0 .../evidence/wp3-forward/readme-ms-agent.md | 0 .../evidence/wp3-forward/readme-storm.md | 0 .../evidence/wp3-forward/report-1280.png | Bin .../evidence/wp3-forward/report-360.png | Bin .../evidence/wp3-forward/report-source.md | 0 .../evidence/wp3-forward/report.html | 0 .../evidence/wp4/pr84-checks-c0f466d2.txt | 0 .../evidence/wp4/pr84-merged.json | 0 .../evidence/wp4/review-verdict.md | 0 .../evidence/wp4/windows-shortname-repro.txt | 0 .../evidence/wp5-delivery-state.json | 0 .../desktop-c795oh4-baseline-doctor.txt | 0 .../wp5-ssh/desktop-c795oh4-hashes.txt | 0 .../evidence/wp5-ssh/macmini-cf-hashes.txt | 0 .../evidence/wp5-ssh/suji-hashes.txt | 0 54 files changed, 18 insertions(+) rename devlog/{_plan => _fin}/260908_narrative_documents/000_plan.md (85%) rename devlog/{_plan => _fin}/260908_narrative_documents/001_sources.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/002_verifiers.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/010_reader_documents.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/020_deep_research.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/030_pr84_integration.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/040_release_0_2_24.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/050_deployment.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/check-links.mjs (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/local-installed-verified.json (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/pr88-checks.txt (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/pr89-checks.txt (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp2-audit/verdict.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp2-forward/fresh-reader.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp2-forward/raw-evidence-dump.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp2-forward/self-check.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp2-forward/status-report.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-audit/verdict.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward-run1/gap-matrix.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward-run1/journal.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward-run1/ledger.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward-run1/plan.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward-run1/report-source.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward-run1/report.html (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward-run2/plan.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/api-assafelovic__gpt-researcher.json (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/api-bytedance__deer-flow.json (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/api-dzhng__deep-research.json (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/api-langchain-ai__deepagents.json (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/api-langchain-ai__open_deep_research.json (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/api-modelscope__ms-agent.json (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/api-stanford-oval__storm.json (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/gap-matrix.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/journal.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/ledger.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/plan.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/readme-deer-flow.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/readme-dzhng-deep-research.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/readme-gpt-researcher.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/readme-ms-agent.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/readme-storm.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/report-1280.png (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/report-360.png (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/report-source.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp3-forward/report.html (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp4/pr84-checks-c0f466d2.txt (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp4/pr84-merged.json (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp4/review-verdict.md (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp4/windows-shortname-repro.txt (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp5-delivery-state.json (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-baseline-doctor.txt (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-hashes.txt (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp5-ssh/macmini-cf-hashes.txt (100%) rename devlog/{_plan => _fin}/260908_narrative_documents/evidence/wp5-ssh/suji-hashes.txt (100%) diff --git a/devlog/_plan/260908_narrative_documents/000_plan.md b/devlog/_fin/260908_narrative_documents/000_plan.md similarity index 85% rename from devlog/_plan/260908_narrative_documents/000_plan.md rename to devlog/_fin/260908_narrative_documents/000_plan.md index d9e4616a..ec98a5bc 100644 --- a/devlog/_plan/260908_narrative_documents/000_plan.md +++ b/devlog/_fin/260908_narrative_documents/000_plan.md @@ -104,3 +104,21 @@ See 001_sources.md for research and 002_verifiers.md for commands with observed - wp5 P (re-entry): 040/050 re-verified against dev c44ab989 (PRs #87 and #84 both in). Version surfaces to bump: package.json, cli, 8 component package.json, gui, lock, plugin.json stamp, inventory.json, CHANGELOG (top Unreleased -> 0.2.24; stale one deleted). +- wp5 D: DONE. v0.2.24 released from main bb852272 (run 34185250992); installed and + verified on local, macmini-cf, suji, desktop-c795oh4 (941/941 payload files each). + +## D closure + +DONE. All five criteria met with captured evidence (`cxc loop validate` OK). What did +not go as planned: two deep-research trial leaves stalled after planning (kept as +run1/run2), the Windows CI on PR #84 exposed an 8.3 short-name path bug, and an +independent review found two Git-environment leaks in the contributor's binding +code; all were fixed with red/green tests before merge. The Windows deployment +script's first regex edit corrupted config.toml and was restored from the preimage. +Follow-ups not done here: parent-directory TOCTOU on `.codexclaw/sources` (accepted +under the documented same-user exclusion), `session-binding.ts` JS realpath +normalization, and the desktop-c795oh4 environment WARNs (Python store alias, codex +features under non-interactive SSH), which predate this release. +What would invalidate this: a fresh session on any target loading a pre-0.2.24 skill +body, a mismatch between the release payload and installed hashes, or a future +forward-use trial that again stalls with zero source opens. diff --git a/devlog/_plan/260908_narrative_documents/001_sources.md b/devlog/_fin/260908_narrative_documents/001_sources.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/001_sources.md rename to devlog/_fin/260908_narrative_documents/001_sources.md diff --git a/devlog/_plan/260908_narrative_documents/002_verifiers.md b/devlog/_fin/260908_narrative_documents/002_verifiers.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/002_verifiers.md rename to devlog/_fin/260908_narrative_documents/002_verifiers.md diff --git a/devlog/_plan/260908_narrative_documents/010_reader_documents.md b/devlog/_fin/260908_narrative_documents/010_reader_documents.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/010_reader_documents.md rename to devlog/_fin/260908_narrative_documents/010_reader_documents.md diff --git a/devlog/_plan/260908_narrative_documents/020_deep_research.md b/devlog/_fin/260908_narrative_documents/020_deep_research.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/020_deep_research.md rename to devlog/_fin/260908_narrative_documents/020_deep_research.md diff --git a/devlog/_plan/260908_narrative_documents/030_pr84_integration.md b/devlog/_fin/260908_narrative_documents/030_pr84_integration.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/030_pr84_integration.md rename to devlog/_fin/260908_narrative_documents/030_pr84_integration.md diff --git a/devlog/_plan/260908_narrative_documents/040_release_0_2_24.md b/devlog/_fin/260908_narrative_documents/040_release_0_2_24.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/040_release_0_2_24.md rename to devlog/_fin/260908_narrative_documents/040_release_0_2_24.md diff --git a/devlog/_plan/260908_narrative_documents/050_deployment.md b/devlog/_fin/260908_narrative_documents/050_deployment.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/050_deployment.md rename to devlog/_fin/260908_narrative_documents/050_deployment.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/check-links.mjs b/devlog/_fin/260908_narrative_documents/evidence/check-links.mjs similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/check-links.mjs rename to devlog/_fin/260908_narrative_documents/evidence/check-links.mjs diff --git a/devlog/_plan/260908_narrative_documents/evidence/local-installed-verified.json b/devlog/_fin/260908_narrative_documents/evidence/local-installed-verified.json similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/local-installed-verified.json rename to devlog/_fin/260908_narrative_documents/evidence/local-installed-verified.json diff --git a/devlog/_plan/260908_narrative_documents/evidence/pr88-checks.txt b/devlog/_fin/260908_narrative_documents/evidence/pr88-checks.txt similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/pr88-checks.txt rename to devlog/_fin/260908_narrative_documents/evidence/pr88-checks.txt diff --git a/devlog/_plan/260908_narrative_documents/evidence/pr89-checks.txt b/devlog/_fin/260908_narrative_documents/evidence/pr89-checks.txt similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/pr89-checks.txt rename to devlog/_fin/260908_narrative_documents/evidence/pr89-checks.txt diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp2-audit/verdict.md b/devlog/_fin/260908_narrative_documents/evidence/wp2-audit/verdict.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp2-audit/verdict.md rename to devlog/_fin/260908_narrative_documents/evidence/wp2-audit/verdict.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp2-forward/fresh-reader.md b/devlog/_fin/260908_narrative_documents/evidence/wp2-forward/fresh-reader.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp2-forward/fresh-reader.md rename to devlog/_fin/260908_narrative_documents/evidence/wp2-forward/fresh-reader.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp2-forward/raw-evidence-dump.md b/devlog/_fin/260908_narrative_documents/evidence/wp2-forward/raw-evidence-dump.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp2-forward/raw-evidence-dump.md rename to devlog/_fin/260908_narrative_documents/evidence/wp2-forward/raw-evidence-dump.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp2-forward/self-check.md b/devlog/_fin/260908_narrative_documents/evidence/wp2-forward/self-check.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp2-forward/self-check.md rename to devlog/_fin/260908_narrative_documents/evidence/wp2-forward/self-check.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp2-forward/status-report.md b/devlog/_fin/260908_narrative_documents/evidence/wp2-forward/status-report.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp2-forward/status-report.md rename to devlog/_fin/260908_narrative_documents/evidence/wp2-forward/status-report.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-audit/verdict.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-audit/verdict.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-audit/verdict.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-audit/verdict.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward-run1/gap-matrix.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward-run1/gap-matrix.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward-run1/gap-matrix.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward-run1/gap-matrix.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward-run1/journal.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward-run1/journal.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward-run1/journal.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward-run1/journal.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward-run1/ledger.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward-run1/ledger.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward-run1/ledger.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward-run1/ledger.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward-run1/plan.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward-run1/plan.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward-run1/plan.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward-run1/plan.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward-run1/report-source.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward-run1/report-source.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward-run1/report-source.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward-run1/report-source.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward-run1/report.html b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward-run1/report.html similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward-run1/report.html rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward-run1/report.html diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward-run2/plan.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward-run2/plan.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward-run2/plan.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward-run2/plan.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/api-assafelovic__gpt-researcher.json b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/api-assafelovic__gpt-researcher.json similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/api-assafelovic__gpt-researcher.json rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/api-assafelovic__gpt-researcher.json diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/api-bytedance__deer-flow.json b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/api-bytedance__deer-flow.json similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/api-bytedance__deer-flow.json rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/api-bytedance__deer-flow.json diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/api-dzhng__deep-research.json b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/api-dzhng__deep-research.json similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/api-dzhng__deep-research.json rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/api-dzhng__deep-research.json diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/api-langchain-ai__deepagents.json b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/api-langchain-ai__deepagents.json similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/api-langchain-ai__deepagents.json rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/api-langchain-ai__deepagents.json diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/api-langchain-ai__open_deep_research.json b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/api-langchain-ai__open_deep_research.json similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/api-langchain-ai__open_deep_research.json rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/api-langchain-ai__open_deep_research.json diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/api-modelscope__ms-agent.json b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/api-modelscope__ms-agent.json similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/api-modelscope__ms-agent.json rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/api-modelscope__ms-agent.json diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/api-stanford-oval__storm.json b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/api-stanford-oval__storm.json similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/api-stanford-oval__storm.json rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/api-stanford-oval__storm.json diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/gap-matrix.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/gap-matrix.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/gap-matrix.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/gap-matrix.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/journal.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/journal.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/journal.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/journal.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/ledger.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/ledger.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/ledger.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/ledger.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/plan.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/plan.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/plan.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/plan.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/readme-deer-flow.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/readme-deer-flow.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/readme-deer-flow.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/readme-deer-flow.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/readme-dzhng-deep-research.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/readme-dzhng-deep-research.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/readme-dzhng-deep-research.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/readme-dzhng-deep-research.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/readme-gpt-researcher.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/readme-gpt-researcher.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/readme-gpt-researcher.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/readme-gpt-researcher.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/readme-ms-agent.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/readme-ms-agent.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/readme-ms-agent.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/readme-ms-agent.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/readme-storm.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/readme-storm.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/readme-storm.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/readme-storm.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/report-1280.png b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/report-1280.png similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/report-1280.png rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/report-1280.png diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/report-360.png b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/report-360.png similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/report-360.png rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/report-360.png diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/report-source.md b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/report-source.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/report-source.md rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/report-source.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp3-forward/report.html b/devlog/_fin/260908_narrative_documents/evidence/wp3-forward/report.html similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp3-forward/report.html rename to devlog/_fin/260908_narrative_documents/evidence/wp3-forward/report.html diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp4/pr84-checks-c0f466d2.txt b/devlog/_fin/260908_narrative_documents/evidence/wp4/pr84-checks-c0f466d2.txt similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp4/pr84-checks-c0f466d2.txt rename to devlog/_fin/260908_narrative_documents/evidence/wp4/pr84-checks-c0f466d2.txt diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp4/pr84-merged.json b/devlog/_fin/260908_narrative_documents/evidence/wp4/pr84-merged.json similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp4/pr84-merged.json rename to devlog/_fin/260908_narrative_documents/evidence/wp4/pr84-merged.json diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp4/review-verdict.md b/devlog/_fin/260908_narrative_documents/evidence/wp4/review-verdict.md similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp4/review-verdict.md rename to devlog/_fin/260908_narrative_documents/evidence/wp4/review-verdict.md diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp4/windows-shortname-repro.txt b/devlog/_fin/260908_narrative_documents/evidence/wp4/windows-shortname-repro.txt similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp4/windows-shortname-repro.txt rename to devlog/_fin/260908_narrative_documents/evidence/wp4/windows-shortname-repro.txt diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp5-delivery-state.json b/devlog/_fin/260908_narrative_documents/evidence/wp5-delivery-state.json similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp5-delivery-state.json rename to devlog/_fin/260908_narrative_documents/evidence/wp5-delivery-state.json diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-baseline-doctor.txt b/devlog/_fin/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-baseline-doctor.txt similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-baseline-doctor.txt rename to devlog/_fin/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-baseline-doctor.txt diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-hashes.txt b/devlog/_fin/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-hashes.txt similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-hashes.txt rename to devlog/_fin/260908_narrative_documents/evidence/wp5-ssh/desktop-c795oh4-hashes.txt diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/macmini-cf-hashes.txt b/devlog/_fin/260908_narrative_documents/evidence/wp5-ssh/macmini-cf-hashes.txt similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/macmini-cf-hashes.txt rename to devlog/_fin/260908_narrative_documents/evidence/wp5-ssh/macmini-cf-hashes.txt diff --git a/devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/suji-hashes.txt b/devlog/_fin/260908_narrative_documents/evidence/wp5-ssh/suji-hashes.txt similarity index 100% rename from devlog/_plan/260908_narrative_documents/evidence/wp5-ssh/suji-hashes.txt rename to devlog/_fin/260908_narrative_documents/evidence/wp5-ssh/suji-hashes.txt From 53029934eb8b07cbbe2618863702477d7d9618e8 Mon Sep 17 00:00:00 2001 From: thisisjun786 <259586770+thisisjun786@users.noreply.github.com> Date: Tue, 8 Sep 2026 05:49:02 +0000 Subject: [PATCH 003/108] fix: make executor canonical across dispatch and exit verification --- CHANGELOG.md | 6 ++ README.ko.md | 14 ++++- README.md | 14 ++++- .../000_plan.md | 28 ++++++++++ .../001_evidence.md | 11 ++++ .../010_executor.md | 29 ++++++++++ plugins/codexclaw/agents/README.md | 55 +++++++++++-------- plugins/codexclaw/agents/executor.toml | 5 +- .../pabcd-state/dist/review-observer.js | 5 +- .../pabcd-state/dist/subagent-evidence.js | 8 +-- .../pabcd-state/src/review-observer.ts | 5 +- .../pabcd-state/src/subagent-evidence.ts | 8 +-- .../pabcd-state/test/review-deadlock.test.ts | 15 +++++ .../test/subagent-evidence.test.ts | 20 ++++--- .../subagent-config/dist/spawn-attach-hook.js | 7 ++- .../subagent-config/dist/spawn-wrapper.js | 10 ++-- .../subagent-config/src/spawn-attach-hook.ts | 7 ++- .../subagent-config/src/spawn-wrapper.ts | 12 ++-- .../test/spawn-attach-hook.test.ts | 25 +++++++++ .../test/spawn-wrapper.test.ts | 6 +- .../subagent-stop-verifying-evidence.json | 2 +- plugins/codexclaw/inventory.json | 2 +- .../skills/pabcd/references/delegation.md | 10 +++- 23 files changed, 231 insertions(+), 73 deletions(-) create mode 100644 devlog/_plan/260908_executor_role_registration/000_plan.md create mode 100644 devlog/_plan/260908_executor_role_registration/001_evidence.md create mode 100644 devlog/_plan/260908_executor_role_registration/010_executor.md diff --git a/CHANGELOG.md b/CHANGELOG.md index 11321c8e..d9c1fae5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,12 @@ All notable changes to codexclaw are documented here. The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and the project uses [semantic versioning](https://semver.org/spec/v2.0.0.html). +## Unreleased + +- Make executor the canonical implementation dispatch role. Add explicit, non-overwriting + `cxc subagents register executor` setup; preserve legacy worker model routing and exit + evidence checks. Start a new session after registration and re-approve changed hooks. + ## [0.2.24] - 2026-09-08 ### Added diff --git a/README.ko.md b/README.ko.md index 0ecb4738..7230faba 100644 --- a/README.ko.md +++ b/README.ko.md @@ -49,13 +49,23 @@ IDLE ── P ── A ── B ── C ── D ── IDLE ## 설치 -두 줄이면 설치 끝. 빌드도, npm install도, 설정 파일 수정도 없다. +플러그인을 설치한 뒤 구현 역할을 한 번 등록한다. 빌드나 npm install은 필요 없다. ```bash codex plugin marketplace add https://github.com/lidge-jun/codexclaw codex plugin add codexclaw@codexclaw ``` +구현 담당 이름은 화면·설정·호출 모두 `executor`다. 설치 경로의 CLI로 등록한다: + +```sh +node "/bin/cxc.mjs" subagents register executor +``` + +새 Codex 세션에서 호출 도구의 역할 목록에 `executor`가 있는지 확인한다. +기존 사용자 역할과 모델 설정은 보존하며, 같은 이름의 다른 파일은 덮어쓰지 않는다. +[역할 등록과 기존 worker 호환 안내](plugins/codexclaw/agents/README.md)를 참고한다. + 설치 후 Codex를 재시작하고 뜨는 승인 창에서 22개 훅을 승인하면 된다(업그레이드 후에도 다시 승인 — 콘텐츠 해시 신뢰 모델). 채팅에서 바로 쓸 수 있고, 터미널 표면도 같이 배송된다 — 페이로드에 자체 `cxc` 디스패처가 들어 있어 에이전트의 `cxc orchestrate` 명령이 모든 설치에서 동작한다: - `orchestrate status` — PABCD 상태 머신 확인 @@ -111,7 +121,7 @@ plugins/codexclaw/ │ ├── pre-tool-use-* skill attach, goal guards, patch lint, interview guard │ ├── post-tool-use-* interview capture, render observation │ ├── stop-* PABCD continuation under active goals -│ ├── subagent-stop-* evidence verification for worker dispatches +│ ├── subagent-stop-* evidence verification for executor dispatches (legacy worker supported) │ └── post-compact-* cursor reinject, recall context, bg-terminal affordance │ ├── components/ 8 isolated feature modules (src + dist) diff --git a/README.md b/README.md index c567ba71..1c4c163d 100644 --- a/README.md +++ b/README.md @@ -49,13 +49,23 @@ IDLE ── P ── A ── B ── C ── D ── IDLE ## Install -2 lines to install. No build step, no npm install, no config edits. +Install the plugin, then register the implementation role once. No build step or npm install. ```bash codex plugin marketplace add https://github.com/lidge-jun/codexclaw codex plugin add codexclaw@codexclaw ``` +Before implementation dispatch, register the canonical executor role once: + +```sh +node "/bin/cxc.mjs" subagents register executor +``` + +Start a new session and verify `executor` appears in the live spawn schema. Registration +preserves existing user roles and project model settings; conflicting files are refused. +See [role setup and legacy worker compatibility](plugins/codexclaw/agents/README.md). + Then restart Codex and approve the 23 hooks when prompted (upgrades ask again — content-hash trust). Everything runs from chat, and the terminal surface ships too — the payload includes its own `cxc` dispatcher, so agent-driven `cxc orchestrate` commands work on every install: - `orchestrate status` — check the PABCD state machine @@ -111,7 +121,7 @@ plugins/codexclaw/ │ ├── pre-tool-use-* skill attach, goal guards, patch lint, interview guard │ ├── post-tool-use-* interview capture, render observation │ ├── stop-* PABCD continuation under active goals -│ ├── subagent-stop-* evidence verification for worker dispatches +│ ├── subagent-stop-* evidence verification for executor dispatches (legacy worker supported) │ └── post-compact-* cursor reinject, recall context, bg-terminal affordance │ ├── components/ 8 isolated feature modules (src + dist) diff --git a/devlog/_plan/260908_executor_role_registration/000_plan.md b/devlog/_plan/260908_executor_role_registration/000_plan.md new file mode 100644 index 00000000..d43b212c --- /dev/null +++ b/devlog/_plan/260908_executor_role_registration/000_plan.md @@ -0,0 +1,28 @@ +# Executor role registration + +CXC displays and stores executor but emits worker and only recognizes worker at its spawn/exit boundaries. Make executor the canonical registered implementation role, retaining worker solely as a legacy input alias. Explicit setup must preserve user configuration and require a new Codex session to discover the role. + +- Class: C4 (role registration and exit verification boundary); one satisfy-spec PABCD cycle. +- Trigger: Jun authorized local implementation, local installation and an ordinary PR. +- Goal: executor remains executor from UI/store through native dispatch and evidence verification. +- Non-goals: no UI redesign, model changes, automatic hook trust, deployment/merge, blanket renaming of generic worker prose or historical records. +- Verifiers: Node tests for subagent-config and affected pabcd-state hooks, shipped CLI temporary-home registration, build/gate, full npm test, native Codex role discovery/live dispatch if the current host can refresh roles. Existing wrapper+CLI tests ran: 37 pass, 0 fail; direct test paths observe the owners. New tests are proposed until implemented. Preserve real-host limitations in delivery. +- Stop: passing review/checks, local patch and role registration verified, PR published; unresolved host discovery is reported, never claimed tested. +- Memory/evidence: this unit and task-local evidence outside tracked source for large logs. +- Outcomes: verified PR plus local install; or explicit blocking finding with implementation preserved. +- Escalation: existing executor collision cannot be overwritten. Two failed delegated attempts return implementation to main; any new delegation scope first amends this plan. + +## Repository and structure +No repository AGENTS.md/POLICY.md found. Follow existing Node24 TS source + generated dist, node:test and numbered devlog conventions. +`subagent-config/{src,test}` owns CLI/store/spawn; `pabcd-state/{src,test}` owns exit evidence; `hooks/` selects exits; `agents/` is the prompt source. `agents/README.md` and README installation are source-of-truth targets. +No new dependency, daemon, automatic mutation hook or framework. Configuration alone cannot fix inferRole(executor) returning explorer; reuse existing CLI and canonical TOML prompt. Add a colocated registration module because safely publishing a user-role file is distinct from routing. + +## Threat boundary +Assets: user role files/model settings and delegated verification. Explicit registration reads only the shipped executor template, writes only CODEX_HOME/agents/executor.toml (default ~/.codex), no model/effort/sandbox/approval overrides. Reject symlink/non-file destinations and symlink agents directory, preserve conflicting existing file, publish without overwrite, repeat exact content idempotently. No role removal or worker deletion. Malicious project text cannot trigger registration; only explicit CLI command can. Local same-user filesystem races are residual risk; do not claim hostile-user isolation. +Guard layer: hook early validation; surface: spawn and SubagentStop. Bypass: disabled/untrusted hooks or direct host call. Residual: host controls actual permissions. Wording: early validation, not unbypassable enforcement; final layer: native Codex permission policy. + +## Delegation +Main owns role mapping, prompt/README/skill call guidance, model routing tests, integration, local patch and PR. During B one worker owns registration module + CLI wiring + their tests only. Independent reviewer audits plan now and another fresh review checks final code. No shared write paths and no child FSM mutations. + +## Previous cycle +Previous cycle fixed source worktree binding and ended IDLE. This distinct cycle fixes role identity; no old worktree/phase evidence is reused as proof. diff --git a/devlog/_plan/260908_executor_role_registration/001_evidence.md b/devlog/_plan/260908_executor_role_registration/001_evidence.md new file mode 100644 index 00000000..ba205986 --- /dev/null +++ b/devlog/_plan/260908_executor_role_registration/001_evidence.md @@ -0,0 +1,11 @@ +# Evidence + +Base dev commit: 6d70ef4. Native session cwd is /home/jun/tmp (not a Git repository); FSM records work here and source proof is taken directly from the isolated worktree, not attributed to the native cwd. + +Plan audit: independent reviewer 01a07f87-2006-7662-959e-6df91491f290, native runtime model anthropic/claude-fable-5-1 / high. First GO-WITH-FIXES (3): hook matcher trust/inventory, actual native role/exit proof, registration prerequisite vs fallback. First two folded; automatic fallback rebutted for canonical naming requirement; reviewer re-audit VERDICT: PASS. + +Baseline: existing wrapper+CLI 37 passed; spawn/exit/review boundaries 143 passed. Direct probe: inferRole(executor, implementation)=explorer; with review word=reviewer; executor not in exit gate. + +RED: new regressions failed before production edits (wrong executor model, wrong role, missing evidence block, executor accepted as review signoff). GREEN: same regressions with affected suites: 176 tests passed, 0 failed. Logs: /home/jun/tmp/cxc-update-20260908-01a07d17/executor-{red,green}.log. + +Temp-home native app-server strict config/read accepted template but returned agents:null. This is NOT proof of native role discovery. Replaced with fresh-session live evidence requirement. diff --git a/devlog/_plan/260908_executor_role_registration/010_executor.md b/devlog/_plan/260908_executor_role_registration/010_executor.md new file mode 100644 index 00000000..6d49a28b --- /dev/null +++ b/devlog/_plan/260908_executor_role_registration/010_executor.md @@ -0,0 +1,29 @@ +# Implementation contract + +Depends on existing RoleName executor; no persisted enum or setting migration. + +## File changes +- NEW subagent-config/src/role-registration.ts and test/role-registration.test.ts: explicit registration API, pinned shipped template path derived from import.meta.url (works src/dist), remove model sentinel from installed role, preserve all instruction text, exclusive publication and idempotence, conflict and symlink errors. CLI `cxc subagents register executor`; no arbitrary role/path/prompt arguments. Home injection only at API for tests or standard CODEX_HOME environment. Output installed path and restart guidance. No config.toml editing. +- MODIFY subagent-config/src/cli.ts and test/cli.test.ts: parse register executor strictly, invoke registration and report error nonzero. CLI does not auto-register during list/get/set/spawn. +- MODIFY subagent-config/src/spawn-wrapper.ts, test/spawn-wrapper.test.ts: executor payload agent_type becomes executor; pure builder remains filesystem-free. Setup prerequisite explicit in docs. Existing worker direct calls remain accepted by hook. +- MODIFY subagent-config/src/spawn-attach-hook.ts, test/spawn-attach-hook.test.ts: executor and worker select executor before keyword inference; explicit reviewer selects reviewer. Preserve existing explorer keyword compatibility, fork restrictions, recursion guard, model/effort override rules. No new message role marker. +- MODIFY pabcd-state/src/subagent-evidence.ts, src/review-observer.ts, test/subagent-evidence.test.ts and test/review-deadlock.test.ts, hooks/subagent-stop-verifying-evidence.json: gate executor and worker identically; both excluded from review observer; matcher ^(executor|worker)$. No permission or evidence relaxation. +- MODIFY agents/executor.toml comments, agents/README.md, README.md, README.ko.md, active skill call examples: canonical executor with explicit registration prerequisite; worker only legacy native compatibility. Historical devlogs untouched. +- GENERATED corresponding dist JS via npm run build; CHANGELOG Unreleased and inventory if required. + +## Chain and acceptance +Creation: registration CLI -> shipped template -> user role TOML named executor. Host reads that role next session; builder emits executor -> spawn hook chooses existing executor settings -> subagent exit matcher/runtime evidence verifies executor. Persistence: roles.executor unchanged. Deserialization: native role loader and existing store unchanged; worker input alias remains. Consumers: spawn wrapper, inferRole, exit matcher, evidence gate and review observer. UI unchanged because already executor. + +1. Fresh temp home: register creates parseable executor TOML, instruction body equals shipped template, no model/effort/sandbox/approval override; repeat unchanged; CLI rejects unknown names/extra args. +2. Existing different file or symlink/directory: nonzero, original bytes unchanged; user config and worker file untouched. +3. executor message mentioning review still selects executor configured model+effort; same behavior for worker on v1/v2 fresh spawn. Full-history fork retains current no-override policy. +4. Executor exit without evidence blocks exactly like worker; reviewer/explorer unaffected; review observer excludes executor even with verdict-looking output. +5. Shipped payload includes new module and CLI works without repo node_modules. A fresh native session must expose executor in its spawn schema and record executor as the spawned role. Capture native SubagentStop evidence (or native runtime logs) to prove exit identity. Config/read agents:null is explicitly NOT discovery proof. If unavailable, report custom-role exit identity unverified; only worker is live-proven. +6. Local apply backs up plugin/runtime files and global role config; uses registration command; change global guidance worker -> executor only in implementation role selection. Preserve legacy worker file and project model settings. +7. PR targets upstream dev from isolated branch; no merge. Full relevant checks + negative cases, retain logs and final review. + +## Audit fold-back +1. Matcher identity changes require hook re-approval. Update plugins/codexclaw/inventory.json via the existing generator. Run cxc doctor after local application; explicitly report any Modified/untrusted hook and require the normal user-facing reapproval. Never hand-edit trust state or claim active hooks from unit tests. Package tests prove matcher selection; native logs prove delivery only when available. +2. Fresh-session native verification replaces config/read as discovery evidence. Capture tool schema/actual agent_role and executor exit behavior, including missing receipt failure. If runtime cannot refresh, local code delivery remains distinct from activation. +3. Executor setup is a deliberate new prerequisite, not an optional hidden fallback: the user explicitly chose one canonical name. Canonical builders emit executor; docs/active delegation instructions require register + new session, then check the actual exposed role before calling. On older/unregistered hosts report the setup requirement; do not invent executor support or silently relabel as worker. Legacy callers that explicitly emit worker continue to work. This rebuts an unconditional automatic worker fallback because it perpetuates the requested inconsistency. Registration is not performed inside a spawn hook. +Inline prompts remain intentionally for per-project promptOverride and old worker callers; native base instructions and inline overrides are not a claim that the native developer instruction is erased. Preserve existing precedence; document this limitation. state.ts missing-agentType legacy fallback remains worker to avoid recategorizing old tombstones; actual executor entries already preserve their string. diff --git a/plugins/codexclaw/agents/README.md b/plugins/codexclaw/agents/README.md index faf14e15..57ca32c9 100644 --- a/plugins/codexclaw/agents/README.md +++ b/plugins/codexclaw/agents/README.md @@ -1,7 +1,7 @@ # codexclaw subagent roles These `.toml` files define codexclaw's subagent roles — the Codex equivalent of orchestrated -"employees". Each role pairs a built-in Codex `agent_type` with a developer prompt that routes +"employees". Each role pairs a native Codex `agent_type` with a developer prompt that routes through the `dev-*` skills for its surface. ## Roles @@ -10,41 +10,50 @@ through the `dev-*` skills for its surface. |------|------------|--------|----------------------------| | `explorer` | `explorer` | no | dev-architecture, dev-debugging, dev-backend/frontend | | `reviewer` | `explorer` | no | dev-code-reviewer, dev-security, dev-architecture, dev-testing | -| `executor` | `worker` | yes (scoped) | dev (classifier) + surface router (frontend/backend/testing/scaffolding) | +| `executor` | `executor` (registered) | yes (scoped) | dev (classifier) + surface router (frontend/backend/testing/scaffolding) | Built-in `agent_type` values are codex-native (`core/src/agent/role.rs`: `default`, `explorer`, `worker`). `explorer` is read-only; `worker` may write. -## Phase 1 = B-opt2 (inline injection) +## Register executor before dispatch -Codex plugin manifests expose only `skills`, `hooks`, `mcpServers`, and `apps` — there is **no -`agents` field** (`core-plugins/src/manifest.rs`). Agent roles are discovered only per config -layer at `/agents/*.toml` (`core/src/config/agent_roles.rs`), and a plugin's -install directory is never registered as a config layer. So these files are **not -auto-registered** as live roles. - -Instead, the main agent injects each role's `developer_instructions` **inline** when spawning: +Plugin directories are not Codex configuration layers, so installing the plugin alone +cannot register these TOML files as live roles. Run the explicit setup command: +```sh +node "/bin/cxc.mjs" subagents register executor ``` -spawn_agent({ agent_type: "explorer", task_name: "explorer_", fork_turns: "none", - message: "TASK: " }) -spawn_agent({ agent_type: "worker", task_name: "executor_", fork_turns: "none", + +This creates `$CODEX_HOME/agents/executor.toml` (default `~/.codex/agents/executor.toml`) +from the shipped executor prompt, omitting the plugin's `model = "default"` sentinel. +The installed role does not override model, effort, sandbox or approval policy. Identical +files are left unchanged; conflicting files and symlinks are refused without overwrite. +Existing worker files and project model settings are preserved. + +Start a new Codex session and check that the live spawn schema exposes `executor`. +If it does not, or the host rejects that agent_type, report the unmet setup prerequisite; +do not invent support or silently switch roles. Registration is never run by a spawn hook. +The canonical builder emits `executor`; callers on older setups can still explicitly use +`worker` as a legacy alias. Both names select `roles.executor` and require exit evidence. + +After upgrading the SubagentStop matcher, re-approve Modified hooks using Codex's normal +hook approval UI and check `cxc doctor`. A passing unit test does not prove hook delivery. + +Fresh V2 call example (use only fields the live tool exposes): + +```js +spawn_agent({ agent_type: "executor", task_name: "executor_change", fork_turns: "none", message: "TASK: " }) -// V2 shape: task_name is required ([a-z0-9_]+); fork_turns "none" keeps -// agent_type/model/effort overrides legal (a full-history fork rejects them). ``` -This is omo's proven pattern. The `.toml` files are the canonical SOURCE of those prompts and -stay B-opt1-ready: if a future codex build supports plugin- or config-layer role registration, -the same files can be copied into a config-layer `agents/` dir with no rewrite. - -Note: these files intentionally carry no `read_only` key — that is not a valid agent-role-file -field (`ConfigToml` uses `deny_unknown_fields`). Read-only intent is encoded in the -`agent_type` mapping and the developer_instructions. +The native role supplies the base developer instructions. Inline task instructions remain +for project prompt overrides and legacy worker calls. A `promptOverride` replaces the +inline template, not the registered native developer instructions; native permissions +and higher-priority instructions still apply. ## Model / prompt override status -`model = "default"` means inherit the parent model. The `.codexclaw/subagents.json` +The shipped TOML `model = "default"` is a plugin sentinel, not a native model name. The `.codexclaw/subagents.json` store, MCP/GUI roundtrip, and `resolveSpawnConfig(cwd, role)` resolver are shipped; S8/S10 tests prove persistence and resolver behavior. The store also carries a per-role `effort` override (codex wire values low/medium/high/xhigh; null = inherit). diff --git a/plugins/codexclaw/agents/executor.toml b/plugins/codexclaw/agents/executor.toml index 51712f3c..afd2a888 100644 --- a/plugins/codexclaw/agents/executor.toml +++ b/plugins/codexclaw/agents/executor.toml @@ -1,7 +1,6 @@ # codexclaw subagent role: executor -# Bounded code changes in a disjoint write scope. Maps to the codex built-in `worker` agent_type -# (writes enabled). Phase 1 (B-opt2): instructions injected INLINE via spawn_agent message; this -# file is the canonical SOURCE, not an auto-registered role. +# Bounded code changes in a disjoint write scope. Register with `cxc subagents register executor`. +# This is the canonical prompt source; plugin installation alone does not register a native role. name = "executor" description = "Implements a bounded, well-scoped code change in a clear write scope and reports the files it touched." nickname_candidates = ["Builder", "Maker", "Hand"] diff --git a/plugins/codexclaw/components/pabcd-state/dist/review-observer.js b/plugins/codexclaw/components/pabcd-state/dist/review-observer.js index aeed2727..eb643195 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/review-observer.js +++ b/plugins/codexclaw/components/pabcd-state/dist/review-observer.js @@ -8,7 +8,7 @@ * subagent actually ends its turn. * * Never blocks. The existing worker evidence gate does the denying (matcher - * `^worker$`); this observer excludes exactly that type, so a read-only audit is + * `^(executor|worker)$`); this observer excludes both implementation types, so a read-only audit is * never held to a receipt it was designed to skip (DISPATCH-AGENT-TYPE-01), and * the two hooks cannot race over the same child. * @@ -38,6 +38,7 @@ * Fail-open on every IO or parse error: a missed recording costs one more audit * round, while a thrown observer would break an unrelated subagent's exit. */ +import { GATED_AGENT_TYPES } from "./subagent-evidence.js"; import { readState } from "./state.js"; import { appendGoalplanLedger, @@ -59,7 +60,7 @@ export function handleReviewObserver(raw ) { if (payload.hook_event_name !== "SubagentStop") return ""; // A worker's exit belongs to the receipt gate; everything else is decided by // the sign-off below. Not "=== explorer": a v1 child arrives as "default". - if (payload.agent_type === "worker") return ""; + if (GATED_AGENT_TYPES.has(payload.agent_type ?? "")) return ""; const { cwd, session_id: sessionId } = payload; if (!cwd || !sessionId) return ""; diff --git a/plugins/codexclaw/components/pabcd-state/dist/subagent-evidence.js b/plugins/codexclaw/components/pabcd-state/dist/subagent-evidence.js index 4860e1ca..6c6eb68e 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/subagent-evidence.js +++ b/plugins/codexclaw/components/pabcd-state/dist/subagent-evidence.js @@ -1,7 +1,7 @@ /** * subagent-evidence.ts — SubagentStop evidence-receipt gate (lazygap_impl 010). * - * A dispatched WRITE/verify subagent (agent_type "worker") cannot "finish" without a + * A dispatched WRITE/verify subagent (agent_type "executor", or legacy "worker") cannot "finish" without a * non-empty evidence receipt under `.codexclaw/evidence/`. Missing/invalid receipt -> * `decision:"block"` with a verifier directive that re-prompts the CHILD (codex-rs * turn.rs:323). After MAX_ATTEMPTS the directive escalates but remains fail-closed; @@ -56,12 +56,12 @@ import { /** * agent_type values this gate refuses to release without a receipt. - * DISPATCH-AGENT-TYPE-01: only "worker" is gated. Read-only audit/research + * DISPATCH-AGENT-TYPE-01: executor and legacy worker are gated. Read-only audit/research * dispatches MUST use agent_type:"explorer" so they bypass both the hook - * manifest matcher (^worker$) and this runtime gate. See + * manifest matcher (^(executor|worker)$) and this runtime gate. See * structure/20_pabcd_dispatch_doctrine.md §3. */ -export const GATED_AGENT_TYPES = new Set (["worker"]); +export const GATED_AGENT_TYPES = new Set (["executor", "worker"]); /** Blocks allowed per agent before the gate terminates and releases. */ export const MAX_ATTEMPTS = 3; diff --git a/plugins/codexclaw/components/pabcd-state/src/review-observer.ts b/plugins/codexclaw/components/pabcd-state/src/review-observer.ts index 7c3958bc..61222c02 100644 --- a/plugins/codexclaw/components/pabcd-state/src/review-observer.ts +++ b/plugins/codexclaw/components/pabcd-state/src/review-observer.ts @@ -8,7 +8,7 @@ * subagent actually ends its turn. * * Never blocks. The existing worker evidence gate does the denying (matcher - * `^worker$`); this observer excludes exactly that type, so a read-only audit is + * `^(executor|worker)$`); this observer excludes both implementation types, so a read-only audit is * never held to a receipt it was designed to skip (DISPATCH-AGENT-TYPE-01), and * the two hooks cannot race over the same child. * @@ -38,6 +38,7 @@ * Fail-open on every IO or parse error: a missed recording costs one more audit * round, while a thrown observer would break an unrelated subagent's exit. */ +import { GATED_AGENT_TYPES } from "./subagent-evidence.ts"; import { readState } from "./state.ts"; import { appendGoalplanLedger, @@ -59,7 +60,7 @@ export function handleReviewObserver(raw: string): string { if (payload.hook_event_name !== "SubagentStop") return ""; // A worker's exit belongs to the receipt gate; everything else is decided by // the sign-off below. Not "=== explorer": a v1 child arrives as "default". - if (payload.agent_type === "worker") return ""; + if (GATED_AGENT_TYPES.has(payload.agent_type ?? "")) return ""; const { cwd, session_id: sessionId } = payload; if (!cwd || !sessionId) return ""; diff --git a/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts b/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts index 02173901..ac91f52a 100644 --- a/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts +++ b/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts @@ -1,7 +1,7 @@ /** * subagent-evidence.ts — SubagentStop evidence-receipt gate (lazygap_impl 010). * - * A dispatched WRITE/verify subagent (agent_type "worker") cannot "finish" without a + * A dispatched WRITE/verify subagent (agent_type "executor", or legacy "worker") cannot "finish" without a * non-empty evidence receipt under `.codexclaw/evidence/`. Missing/invalid receipt -> * `decision:"block"` with a verifier directive that re-prompts the CHILD (codex-rs * turn.rs:323). After MAX_ATTEMPTS the directive escalates but remains fail-closed; @@ -56,12 +56,12 @@ import type { SubagentStopPayload } from "./hook.ts"; /** * agent_type values this gate refuses to release without a receipt. - * DISPATCH-AGENT-TYPE-01: only "worker" is gated. Read-only audit/research + * DISPATCH-AGENT-TYPE-01: executor and legacy worker are gated. Read-only audit/research * dispatches MUST use agent_type:"explorer" so they bypass both the hook - * manifest matcher (^worker$) and this runtime gate. See + * manifest matcher (^(executor|worker)$) and this runtime gate. See * structure/20_pabcd_dispatch_doctrine.md §3. */ -export const GATED_AGENT_TYPES = new Set(["worker"]); +export const GATED_AGENT_TYPES = new Set(["executor", "worker"]); /** Blocks allowed per agent before the gate terminates and releases. */ export const MAX_ATTEMPTS = 3; diff --git a/plugins/codexclaw/components/pabcd-state/test/review-deadlock.test.ts b/plugins/codexclaw/components/pabcd-state/test/review-deadlock.test.ts index 2aca4cb1..fd389a3b 100644 --- a/plugins/codexclaw/components/pabcd-state/test/review-deadlock.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/review-deadlock.test.ts @@ -328,3 +328,18 @@ test("an approved round cannot be spent after the plan changed", () => { assert.match(res.output, /the plan changed/); } finally { rmSync(cwd, { recursive: true, force: true }); } }); + +test("an executor exit is never recorded by the review observer", () => { + const { cwd, slug } = seedAtA("ex"); + try { + const launchId = openRoundFor(cwd, "ex"); + handleReviewObserver(JSON.stringify({ + hook_event_name: "SubagentStop", session_id: "ex", cwd, + agent_type: "executor", agent_id: "w1", + last_assistant_message: `done\n\nLAUNCH: ${launchId}\nVERDICT: PASS`, + })); + + const round = latestRound(readGoalplan(cwd, slug)!, "plan_audit")!; + assert.equal(round.status, "in_flight", "the receipt gate owns an executor exit"); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); diff --git a/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts b/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts index c91ed8f5..27c7f8b7 100644 --- a/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts @@ -736,8 +736,8 @@ test("010: transcriptHasContextPressure is false for missing/empty path", () => // --- DISPATCH-AGENT-TYPE-01 invariant tests --- -test("DISPATCH-AGENT-TYPE-01: hook manifest matcher gates only worker agents", async () => { - // The SubagentStop hook JSON must match only "worker" so explorer/default agents +test("DISPATCH-AGENT-TYPE-01: hook manifest matcher gates executor and legacy worker agents", async () => { + // The SubagentStop hook JSON must match implementation roles so explorer/default agents // never even trigger the evidence-receipt gate command. This is the first line of // defense; GATED_AGENT_TYPES in the runtime is the second. const { readFileSync } = await import("node:fs"); @@ -750,14 +750,14 @@ test("DISPATCH-AGENT-TYPE-01: hook manifest matcher gates only worker agents", a const matchers = manifest.hooks.SubagentStop.map( (entry: { matcher?: string }) => entry.matcher, ); - // Exactly one entry with the ^worker$ matcher. - assert.deepEqual(matchers, ["^worker$"]); + // Canonical and legacy names share one handler. + assert.deepEqual(matchers, ["^(executor|worker)$"]); }); -test("DISPATCH-AGENT-TYPE-01: GATED_AGENT_TYPES contains only worker", () => { - // Runtime defense-in-depth: even if the hook matcher is changed, only worker +test("DISPATCH-AGENT-TYPE-01: GATED_AGENT_TYPES contains executor and legacy worker", () => { + // Runtime defense-in-depth: even if the hook matcher is changed, only implementation // agents are evidence-gated. Adding a new gated type requires deliberate change. - assert.deepEqual([...GATED_AGENT_TYPES].sort(), ["worker"]); + assert.deepEqual([...GATED_AGENT_TYPES].sort(), ["executor", "worker"]); }); test("DISPATCH-AGENT-TYPE-01: default agent_type is not gated", () => { @@ -817,3 +817,9 @@ test("DISPATCH-AGENT-TYPE-01: worker without token still blocks", () => { const parsed = JSON.parse(out); assert.equal(parsed.decision, "block", "write task should still be gated"); }); + +test("canonical executor exit without a receipt is blocked", () => { + const cwd = tmp(); + const out = runSubagentStopGate(payload(cwd, { agent_type: "executor" })); + assert.equal(JSON.parse(out).decision, "block"); +}); diff --git a/plugins/codexclaw/components/subagent-config/dist/spawn-attach-hook.js b/plugins/codexclaw/components/subagent-config/dist/spawn-attach-hook.js index d04be79a..c0e7c442 100644 --- a/plugins/codexclaw/components/subagent-config/dist/spawn-attach-hook.js +++ b/plugins/codexclaw/components/subagent-config/dist/spawn-attach-hook.js @@ -431,14 +431,15 @@ const REVIEW_KEYWORDS = [ /** * Map the spawn's agent_type (+ message intent) back to a base RoleName. - * explorer/reviewer both spawn as agent_type "explorer"; executor as "worker". - * The agent_type alone cannot tell reviewer from explorer, so review-intent + * executor is canonical; worker is a legacy implementation-role alias. + * Legacy explorer-typed reviews cannot be distinguished by type, so review-intent * keywords in the message upgrade the explorer surface to "reviewer" — this is * what lets a reviewer-specific model in .codexclaw/subagents.json take effect * on hook-path dispatches. */ export function inferRole(agentType , message ) { - if (agentType === "worker") return "executor"; + if (agentType === "executor" || agentType === "worker") return "executor"; + if (agentType === "reviewer") return "reviewer"; const m = (message ?? "").toLowerCase(); return REVIEW_KEYWORDS.some((k) => m.includes(k)) ? "reviewer" : "explorer"; } diff --git a/plugins/codexclaw/components/subagent-config/dist/spawn-wrapper.js b/plugins/codexclaw/components/subagent-config/dist/spawn-wrapper.js index d9065963..12a9ad50 100644 --- a/plugins/codexclaw/components/subagent-config/dist/spawn-wrapper.js +++ b/plugins/codexclaw/components/subagent-config/dist/spawn-wrapper.js @@ -6,8 +6,8 @@ * L9 gap where the resolver existed but nothing consumed it at spawn time. * * Contract (omo B-opt2 parity, agents/README.md): - * - role -> built-in agent_type (explorer/reviewer -> "explorer", executor -> "worker"); - * the wrapper NEVER invents a role name (codex plugins can't register roles). + * - role -> built-in agent_type (explorer/reviewer -> "explorer", executor -> registered "executor"); + * executor requires `cxc subagents register executor` and a fresh Codex session. * - the role prompt is injected INLINE in the message ("TASK: ..."), since plugin * install dirs are not a config layer. * - model selection is not emitted by the v2 builder. The durable per-role model in @@ -20,11 +20,11 @@ import { existsSync, readFileSync, realpathSync } from "node:fs"; import { join, isAbsolute, resolve as resolvePath } from "node:path"; import { resolveSpawnConfig, } from "./store.js"; -/** Built-in codex agent_type each canonical role maps to (core/src/agent/role.rs). */ -export const ROLE_AGENT_TYPE = { +/** Native agent_type for each role; executor is explicitly registered during setup. */ +export const ROLE_AGENT_TYPE = { explorer: "explorer", reviewer: "explorer", - executor: "worker", + executor: "executor", }; /** diff --git a/plugins/codexclaw/components/subagent-config/src/spawn-attach-hook.ts b/plugins/codexclaw/components/subagent-config/src/spawn-attach-hook.ts index 95c756d9..97870255 100644 --- a/plugins/codexclaw/components/subagent-config/src/spawn-attach-hook.ts +++ b/plugins/codexclaw/components/subagent-config/src/spawn-attach-hook.ts @@ -431,14 +431,15 @@ const REVIEW_KEYWORDS = [ /** * Map the spawn's agent_type (+ message intent) back to a base RoleName. - * explorer/reviewer both spawn as agent_type "explorer"; executor as "worker". - * The agent_type alone cannot tell reviewer from explorer, so review-intent + * executor is canonical; worker is a legacy implementation-role alias. + * Legacy explorer-typed reviews cannot be distinguished by type, so review-intent * keywords in the message upgrade the explorer surface to "reviewer" — this is * what lets a reviewer-specific model in .codexclaw/subagents.json take effect * on hook-path dispatches. */ export function inferRole(agentType: unknown, message: string): RoleName { - if (agentType === "worker") return "executor"; + if (agentType === "executor" || agentType === "worker") return "executor"; + if (agentType === "reviewer") return "reviewer"; const m = (message ?? "").toLowerCase(); return REVIEW_KEYWORDS.some((k) => m.includes(k)) ? "reviewer" : "explorer"; } diff --git a/plugins/codexclaw/components/subagent-config/src/spawn-wrapper.ts b/plugins/codexclaw/components/subagent-config/src/spawn-wrapper.ts index 30b9d514..1d76302b 100644 --- a/plugins/codexclaw/components/subagent-config/src/spawn-wrapper.ts +++ b/plugins/codexclaw/components/subagent-config/src/spawn-wrapper.ts @@ -6,8 +6,8 @@ * L9 gap where the resolver existed but nothing consumed it at spawn time. * * Contract (omo B-opt2 parity, agents/README.md): - * - role -> built-in agent_type (explorer/reviewer -> "explorer", executor -> "worker"); - * the wrapper NEVER invents a role name (codex plugins can't register roles). + * - role -> built-in agent_type (explorer/reviewer -> "explorer", executor -> registered "executor"); + * executor requires `cxc subagents register executor` and a fresh Codex session. * - the role prompt is injected INLINE in the message ("TASK: ..."), since plugin * install dirs are not a config layer. * - model selection is not emitted by the v2 builder. The durable per-role model in @@ -20,11 +20,11 @@ import { existsSync, readFileSync, realpathSync } from "node:fs"; import { join, isAbsolute, resolve as resolvePath } from "node:path"; import { resolveSpawnConfig, type RoleName, type SpawnResolution } from "./store.ts"; -/** Built-in codex agent_type each canonical role maps to (core/src/agent/role.rs). */ -export const ROLE_AGENT_TYPE: Record = { +/** Native agent_type for each role; executor is explicitly registered during setup. */ +export const ROLE_AGENT_TYPE: Record = { explorer: "explorer", reviewer: "explorer", - executor: "worker", + executor: "executor", }; /** @@ -335,7 +335,7 @@ export function taskNameForRole(role: RoleName, task: string): string { * full-history fork. `fork_turns: "none"` here keeps that injection legal on V2. */ export interface SpawnPayload { - agent_type: "explorer" | "worker"; + agent_type: "explorer" | "executor"; message: string; /** v2 spawn schema: required task name, `[a-z0-9_]+`. Present when V2 is active. */ task_name?: string; diff --git a/plugins/codexclaw/components/subagent-config/test/spawn-attach-hook.test.ts b/plugins/codexclaw/components/subagent-config/test/spawn-attach-hook.test.ts index 1657d80f..75a5a03b 100644 --- a/plugins/codexclaw/components/subagent-config/test/spawn-attach-hook.test.ts +++ b/plugins/codexclaw/components/subagent-config/test/spawn-attach-hook.test.ts @@ -1082,3 +1082,28 @@ test("BUG-R1: CLI source and dist drain large rewritten spawn JSON over a pipe", rmSync(cwd, { recursive: true, force: true }); } }); + +// Canonical role regression: implementation intent must not become review routing. +test("executor and legacy worker retain executor model and effort on fresh v1/v2 spawns", (t) => { + const cwd = workspaceWithConfig({ + explorer: { mode: "model", model: "explorer-model", effort: "low", promptOverride: null }, + reviewer: { mode: "model", model: "reviewer-model", effort: "medium", promptOverride: null }, + executor: { mode: "model", model: "executor-model", effort: "high", promptOverride: null }, + }); + t.after(() => rmSync(cwd, { recursive: true, force: true })); + for (const agent_type of ["executor", "worker"]) { + for (const surface of [{ fork_context: false }, { task_name: "implement", fork_turns: "none" }]) { + const ui = updatedInputOf(runSpawnAttachHook(spawnPayloadAt(cwd, { + ...surface, agent_type, message: "Implement and review the bounded change", + }))); + assert.equal(ui.agent_type, agent_type); + assert.equal(ui.model, "executor-model"); + assert.equal(ui.reasoning_effort, "high"); + } + } +}); + +test("explicit executor and reviewer roles take precedence over message keywords", () => { + assert.equal(inferRole("executor", "review the implementation"), "executor"); + assert.equal(inferRole("reviewer", "inspect correctness"), "reviewer"); +}); diff --git a/plugins/codexclaw/components/subagent-config/test/spawn-wrapper.test.ts b/plugins/codexclaw/components/subagent-config/test/spawn-wrapper.test.ts index b0598f86..d064d229 100644 --- a/plugins/codexclaw/components/subagent-config/test/spawn-wrapper.test.ts +++ b/plugins/codexclaw/components/subagent-config/test/spawn-wrapper.test.ts @@ -36,10 +36,10 @@ function tmp() { return mkdtempSync(join(tmpdir(), "cxc-spawn-")); } -test("agent_type mapping: explorer/reviewer -> explorer, executor -> worker", () => { +test("agent_type mapping: explorer/reviewer -> explorer, executor -> registered executor", () => { assert.equal(ROLE_AGENT_TYPE.explorer, "explorer"); assert.equal(ROLE_AGENT_TYPE.reviewer, "explorer"); - assert.equal(ROLE_AGENT_TYPE.executor, "worker"); + assert.equal(ROLE_AGENT_TYPE.executor, "executor"); }); test("parseRoleToml reads model sentinel + triple-quoted developer_instructions", () => { @@ -120,7 +120,7 @@ test("buildSpawnPayload: promptOverride REPLACES the TOML developer_instructions resolution: { role: "executor", model: null, usesMainModel: true, effort: null, promptOverride: "CUSTOM PROMPT" }, developerInstructions: "Role: executor (TOML body).", }); - assert.equal(payload.agent_type, "worker"); + assert.equal(payload.agent_type, "executor"); assert.match(payload.message, /CUSTOM PROMPT/); assert.ok(!payload.message.includes("TOML body"), "override must replace the TOML body"); }); diff --git a/plugins/codexclaw/hooks/subagent-stop-verifying-evidence.json b/plugins/codexclaw/hooks/subagent-stop-verifying-evidence.json index 44210d20..24ad5ce6 100644 --- a/plugins/codexclaw/hooks/subagent-stop-verifying-evidence.json +++ b/plugins/codexclaw/hooks/subagent-stop-verifying-evidence.json @@ -10,7 +10,7 @@ "statusMessage": "(codexclaw) Verifying subagent evidence" } ], - "matcher": "^worker$" + "matcher": "^(executor|worker)$" } ] } diff --git a/plugins/codexclaw/inventory.json b/plugins/codexclaw/inventory.json index 783cde44..c107dca6 100644 --- a/plugins/codexclaw/inventory.json +++ b/plugins/codexclaw/inventory.json @@ -238,7 +238,7 @@ "file": "subagent-stop-verifying-evidence.json", "event": "SubagentStop", "component": "pabcd-state", - "matcher": "^worker$" + "matcher": "^(executor|worker)$" }, { "file": "user-prompt-submit-checking-pabcd-trigger.json", diff --git a/plugins/codexclaw/skills/pabcd/references/delegation.md b/plugins/codexclaw/skills/pabcd/references/delegation.md index ec6a002c..b27840e5 100644 --- a/plugins/codexclaw/skills/pabcd/references/delegation.md +++ b/plugins/codexclaw/skills/pabcd/references/delegation.md @@ -1,7 +1,7 @@ ## Delegation Model (subagents) The main session owns the plan, host goal, and every PABCD transition. -At A, dispatch an independent `explorer`; use a `worker` for bounded writes +At A, dispatch an independent `explorer`; use the registered `executor` for bounded writes (DISPATCH-AGENT-TYPE-01). Subagents are leaves (LEAF-TOPOLOGY-01) unless recursion is explicitly granted. Every dispatch carries a structured TASK packet (DISPATCH-TASK-01): @@ -19,7 +19,7 @@ required task source still must be loaded or reported missing before its governe ### Live tool schema and role transport Use the loaded native tool schema, not a version label, to choose arguments. -`explorer`/`worker` express the intended role; `agent_type` and `task_name` are +`explorer`/`executor` express the intended role; `agent_type` and `task_name` are not universal fields. Use them only when exposed. Otherwise put the logical role, task/lens name and exact read/write scope in the task message, without inventing arguments or claiming a native permission profile was selected. @@ -27,6 +27,12 @@ Prompt labels are not enforcement and cannot bypass an actual worker receipt requirement or other runtime guard. If the requested protection cannot be represented, report that gap rather than silently weakening it. +Before executor dispatch, run the authorized setup `cxc subagents register executor` +once, start a new Codex session, and inspect the live spawn schema's role list. If +`executor` is absent or rejected, report the registration/restart prerequisite; do +not silently substitute a role. `worker` is accepted only for legacy callers. +The registration command does not overwrite user roles or alter model/permissions. + Map each logical task to the handle actually returned by the tool: for example, a V1 agent_id or a V2 canonical task_name. Use the actual handle and supported follow-up/wait/retirement schema, never a display label or guessed ID. Apply only From 2f8a13745c8e86f71307b17f12ed0d08d36879d1 Mon Sep 17 00:00:00 2001 From: thisisjun786 <259586770+thisisjun786@users.noreply.github.com> Date: Tue, 8 Sep 2026 05:56:08 +0000 Subject: [PATCH 004/108] feat: register executor explicitly without overwriting user roles --- .../001_evidence.md | 2 + .../010_executor.md | 2 + .../cxc-ops/test/hook-trust.test.ts | 6 +- .../components/subagent-config/dist/cli.js | 16 ++++ .../subagent-config/dist/role-registration.js | 48 +++++++++++ .../components/subagent-config/src/cli.ts | 18 ++++- .../subagent-config/src/role-registration.ts | 48 +++++++++++ .../test/role-registration.test.ts | 79 +++++++++++++++++++ plugins/codexclaw/test/hook-e2e.test.mjs | 6 +- 9 files changed, 218 insertions(+), 7 deletions(-) create mode 100644 plugins/codexclaw/components/subagent-config/dist/role-registration.js create mode 100644 plugins/codexclaw/components/subagent-config/src/role-registration.ts create mode 100644 plugins/codexclaw/components/subagent-config/test/role-registration.test.ts diff --git a/devlog/_plan/260908_executor_role_registration/001_evidence.md b/devlog/_plan/260908_executor_role_registration/001_evidence.md index ba205986..2e000e62 100644 --- a/devlog/_plan/260908_executor_role_registration/001_evidence.md +++ b/devlog/_plan/260908_executor_role_registration/001_evidence.md @@ -9,3 +9,5 @@ Baseline: existing wrapper+CLI 37 passed; spawn/exit/review boundaries 143 passe RED: new regressions failed before production edits (wrong executor model, wrong role, missing evidence block, executor accepted as review signoff). GREEN: same regressions with affected suites: 176 tests passed, 0 failed. Logs: /home/jun/tmp/cxc-update-20260908-01a07d17/executor-{red,green}.log. Temp-home native app-server strict config/read accepted template but returned agents:null. This is NOT proof of native role discovery. Replaced with fresh-session live evidence requirement. + +Implementation delegation: worker 01a07f8d-8805-7d33-addc-d475379dbf97 spent approximately eight minutes investigating without edits. Closed and confirmed no output files; main reclaimed the now-blocking small registration slice rather than dispatching another idle-dependent lane (host critical-path preference). No worker result claimed as implementation evidence. diff --git a/devlog/_plan/260908_executor_role_registration/010_executor.md b/devlog/_plan/260908_executor_role_registration/010_executor.md index 6d49a28b..2a18d2a5 100644 --- a/devlog/_plan/260908_executor_role_registration/010_executor.md +++ b/devlog/_plan/260908_executor_role_registration/010_executor.md @@ -27,3 +27,5 @@ Creation: registration CLI -> shipped template -> user role TOML named executor. 2. Fresh-session native verification replaces config/read as discovery evidence. Capture tool schema/actual agent_role and executor exit behavior, including missing receipt failure. If runtime cannot refresh, local code delivery remains distinct from activation. 3. Executor setup is a deliberate new prerequisite, not an optional hidden fallback: the user explicitly chose one canonical name. Canonical builders emit executor; docs/active delegation instructions require register + new session, then check the actual exposed role before calling. On older/unregistered hosts report the setup requirement; do not invent executor support or silently relabel as worker. Legacy callers that explicitly emit worker continue to work. This rebuts an unconditional automatic worker fallback because it perpetuates the requested inconsistency. Registration is not performed inside a spawn hook. Inline prompts remain intentionally for per-project promptOverride and old worker callers; native base instructions and inline overrides are not a claim that the native developer instruction is erased. Preserve existing precedence; document this limitation. state.ts missing-agentType legacy fallback remains worker to avoid recategorizing old tombstones; actual executor entries already preserve their string. + +Full-suite follow-up: update cxc-ops/test/hook-trust.test.ts live matcher golden fixture. Compute new digest independently using Python sorted JSON + hashlib; existing worker matcher identity is changed deliberately and requires reapproval. GUI router baseline required npm ci; no GUI product edits. diff --git a/plugins/codexclaw/components/cxc-ops/test/hook-trust.test.ts b/plugins/codexclaw/components/cxc-ops/test/hook-trust.test.ts index 95e293b7..6082f949 100644 --- a/plugins/codexclaw/components/cxc-ops/test/hook-trust.test.ts +++ b/plugins/codexclaw/components/cxc-ops/test/hook-trust.test.ts @@ -67,11 +67,11 @@ test("identityHash keeps the matcher in the live SubagentStop hook golden fixtur const document = JSON.parse(readFileSync(fixturePath, "utf8")) as { hooks: { SubagentStop: Array<{ matcher?: string; hooks: HookHandler[] }> }; }; - const group = document.hooks.SubagentStop.find((candidate) => candidate.matcher === "^worker$"); - assert.ok(group, "the real SubagentStop hook must keep its ^worker$ matcher group"); + const group = document.hooks.SubagentStop.find((candidate) => candidate.matcher === "^(executor|worker)$"); + assert.ok(group, "the real SubagentStop hook must keep its executor/worker matcher group"); assert.equal( identityHash("SubagentStop", group.matcher, group.hooks[0]), - "sha256:9afd7aeccc4c240163001eec376823a6566ce30572fafcc300ceb1d6bb4c6290", + "sha256:84e1bb4945bc1f0c31d20ff1cfd3dc555eb266362cc159182962fc6c71f2e6d8", ); }); diff --git a/plugins/codexclaw/components/subagent-config/dist/cli.js b/plugins/codexclaw/components/subagent-config/dist/cli.js index b65de179..ba5e9ecc 100644 --- a/plugins/codexclaw/components/subagent-config/dist/cli.js +++ b/plugins/codexclaw/components/subagent-config/dist/cli.js @@ -14,6 +14,7 @@ * [--prompt |--clear-prompt] */ import { readConfig, setRole, projectConfigTrustToken, ROLES, EFFORTS, } from "./store.js"; +import { registerExecutor } from "./role-registration.js"; import { realpathSync } from "node:fs"; import { fileURLToPath } from "node:url"; @@ -35,6 +36,12 @@ export function parseSubagentsArgs(argv ) { if (sub === "help" || sub === "--help" || sub === "-h") return { action: "help" }; if (sub === "trust-token") return { action: "trust-token" }; + if (sub === "register") { + return argv.length === 2 && argv[1] === "executor" + ? { action: "register", role: "executor" } + : { action: "register", error: "usage: subagents register executor" }; + } + if (sub === "get") { if (!isRole(argv[1])) return { action: "get", error: `unknown role '${argv[1] ?? ""}' (expected ${ROLES.join("|")})` }; return { action: "get", role: argv[1] }; @@ -83,6 +90,7 @@ const HELP = [ " subagents list all role configs", " subagents get show one role config", " subagents set --mode default|model [--model ] [--effort |--clear-effort] [--prompt |--clear-prompt]", + " subagents register executor register native role; restart Codex afterward", " subagents trust-token print an export bound to this repo and exact config", "", ` roles: ${ROLES.join(", ")}`, @@ -111,6 +119,14 @@ export function runSubagents(parsed , cwd ) if (!token) return { code: 1, output: "subagents: cannot hash .codexclaw/subagents.json" }; return { code: 0, output: `export CODEXCLAW_TRUST_PROJECT_SUBAGENTS='${token}'` }; } + case "register": { + try { + const result = registerExecutor(); + return { code: 0, output: `${result.created ? "Registered" : "Already registered"}: ${result.path}\nStart a new Codex session and verify executor appears in the live spawn schema.` }; + } catch (err) { + return { code: 1, output: `subagents: ${err instanceof Error ? err.message : String(err)}` }; + } + } case "set": { try { const cfg = setRole(cwd, parsed.role , parsed.patch ?? {}); diff --git a/plugins/codexclaw/components/subagent-config/dist/role-registration.js b/plugins/codexclaw/components/subagent-config/dist/role-registration.js new file mode 100644 index 00000000..f5d6cbd2 --- /dev/null +++ b/plugins/codexclaw/components/subagent-config/dist/role-registration.js @@ -0,0 +1,48 @@ +/** Explicit native executor registration. Never invoked by hooks or dispatch. */ +import { closeSync, constants, fstatSync, linkSync, lstatSync, mkdirSync, openSync, readFileSync, unlinkSync, writeFileSync } from "node:fs"; +import { homedir } from "node:os"; +import { dirname, join, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; +import { randomUUID } from "node:crypto"; + +function existingRole(path ) { + let fd ; + try { + const st = lstatSync(path); + if (!st.isFile() || st.isSymbolicLink()) throw new Error(`Refusing non-regular role file: ${path}`); + fd = openSync(path, constants.O_RDONLY | constants.O_NOFOLLOW | constants.O_NONBLOCK); + } catch (err) { + if ((err ).code === "ENOENT") return null; + throw err; + } + try { + if (!fstatSync(fd).isFile()) throw new Error(`Refusing non-regular role file: ${path}`); + return readFileSync(fd, "utf8"); + } finally { closeSync(fd); } +} + +export function registerExecutor(codexHome = process.env.CODEX_HOME || join(homedir(), ".codex")) { + const template = resolve(dirname(fileURLToPath(import.meta.url)), "../../../agents/executor.toml"); + const content = readFileSync(template, "utf8").replace(/^model\s*=\s*"default"[^\r\n]*\r?\n/m, ""); + const directory = join(codexHome, "agents"); + mkdirSync(codexHome, { recursive: true }); + try { mkdirSync(directory); } + catch (err) { if ((err ).code !== "EEXIST") throw err; } + const st = lstatSync(directory); + if (!st.isDirectory() || st.isSymbolicLink()) throw new Error(`Refusing non-regular agents directory: ${directory}`); + const path = join(directory, "executor.toml"); + const existing = existingRole(path); + if (existing === content) return { path, created: false }; + if (existing !== null) throw new Error(`Existing executor role differs; preserved ${path}. Compare it with ${template} before updating.`); + const temporary = join(directory, `.executor-${randomUUID()}.tmp`); + writeFileSync(temporary, content, { flag: "wx", mode: 0o600 }); + try { + // Publish complete content without replacing an existing name, including races. + try { linkSync(temporary, path); } + catch (err) { + if ((err ).code === "EEXIST" && existingRole(path) === content) return { path, created: false }; + throw err; + } + } finally { unlinkSync(temporary); } + return { path, created: true }; +} diff --git a/plugins/codexclaw/components/subagent-config/src/cli.ts b/plugins/codexclaw/components/subagent-config/src/cli.ts index 806a2a38..f8b2be34 100644 --- a/plugins/codexclaw/components/subagent-config/src/cli.ts +++ b/plugins/codexclaw/components/subagent-config/src/cli.ts @@ -14,11 +14,12 @@ * [--prompt |--clear-prompt] */ import { readConfig, setRole, projectConfigTrustToken, ROLES, EFFORTS, type RoleName, type RoleConfig, type EffortName } from "./store.ts"; +import { registerExecutor } from "./role-registration.ts"; import { realpathSync } from "node:fs"; import { fileURLToPath } from "node:url"; export interface ParsedSubagentsArgs { - action: "list" | "get" | "set" | "trust-token" | "help"; + action: "list" | "get" | "set" | "trust-token" | "register" | "help"; role?: RoleName; patch?: Partial; error?: string; @@ -35,6 +36,12 @@ export function parseSubagentsArgs(argv: string[]): ParsedSubagentsArgs { if (sub === "help" || sub === "--help" || sub === "-h") return { action: "help" }; if (sub === "trust-token") return { action: "trust-token" }; + if (sub === "register") { + return argv.length === 2 && argv[1] === "executor" + ? { action: "register", role: "executor" } + : { action: "register", error: "usage: subagents register executor" }; + } + if (sub === "get") { if (!isRole(argv[1])) return { action: "get", error: `unknown role '${argv[1] ?? ""}' (expected ${ROLES.join("|")})` }; return { action: "get", role: argv[1] }; @@ -83,6 +90,7 @@ const HELP = [ " subagents list all role configs", " subagents get show one role config", " subagents set --mode default|model [--model ] [--effort |--clear-effort] [--prompt |--clear-prompt]", + " subagents register executor register native role; restart Codex afterward", " subagents trust-token print an export bound to this repo and exact config", "", ` roles: ${ROLES.join(", ")}`, @@ -111,6 +119,14 @@ export function runSubagents(parsed: ParsedSubagentsArgs, cwd: string): Subagent if (!token) return { code: 1, output: "subagents: cannot hash .codexclaw/subagents.json" }; return { code: 0, output: `export CODEXCLAW_TRUST_PROJECT_SUBAGENTS='${token}'` }; } + case "register": { + try { + const result = registerExecutor(); + return { code: 0, output: `${result.created ? "Registered" : "Already registered"}: ${result.path}\nStart a new Codex session and verify executor appears in the live spawn schema.` }; + } catch (err) { + return { code: 1, output: `subagents: ${err instanceof Error ? err.message : String(err)}` }; + } + } case "set": { try { const cfg = setRole(cwd, parsed.role as RoleName, parsed.patch ?? {}); diff --git a/plugins/codexclaw/components/subagent-config/src/role-registration.ts b/plugins/codexclaw/components/subagent-config/src/role-registration.ts new file mode 100644 index 00000000..59c13122 --- /dev/null +++ b/plugins/codexclaw/components/subagent-config/src/role-registration.ts @@ -0,0 +1,48 @@ +/** Explicit native executor registration. Never invoked by hooks or dispatch. */ +import { closeSync, constants, fstatSync, linkSync, lstatSync, mkdirSync, openSync, readFileSync, unlinkSync, writeFileSync } from "node:fs"; +import { homedir } from "node:os"; +import { dirname, join, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; +import { randomUUID } from "node:crypto"; + +function existingRole(path: string): string | null { + let fd: number; + try { + const st = lstatSync(path); + if (!st.isFile() || st.isSymbolicLink()) throw new Error(`Refusing non-regular role file: ${path}`); + fd = openSync(path, constants.O_RDONLY | constants.O_NOFOLLOW | constants.O_NONBLOCK); + } catch (err) { + if ((err as NodeJS.ErrnoException).code === "ENOENT") return null; + throw err; + } + try { + if (!fstatSync(fd).isFile()) throw new Error(`Refusing non-regular role file: ${path}`); + return readFileSync(fd, "utf8"); + } finally { closeSync(fd); } +} + +export function registerExecutor(codexHome = process.env.CODEX_HOME || join(homedir(), ".codex")): { path: string; created: boolean } { + const template = resolve(dirname(fileURLToPath(import.meta.url)), "../../../agents/executor.toml"); + const content = readFileSync(template, "utf8").replace(/^model\s*=\s*"default"[^\r\n]*\r?\n/m, ""); + const directory = join(codexHome, "agents"); + mkdirSync(codexHome, { recursive: true }); + try { mkdirSync(directory); } + catch (err) { if ((err as NodeJS.ErrnoException).code !== "EEXIST") throw err; } + const st = lstatSync(directory); + if (!st.isDirectory() || st.isSymbolicLink()) throw new Error(`Refusing non-regular agents directory: ${directory}`); + const path = join(directory, "executor.toml"); + const existing = existingRole(path); + if (existing === content) return { path, created: false }; + if (existing !== null) throw new Error(`Existing executor role differs; preserved ${path}. Compare it with ${template} before updating.`); + const temporary = join(directory, `.executor-${randomUUID()}.tmp`); + writeFileSync(temporary, content, { flag: "wx", mode: 0o600 }); + try { + // Publish complete content without replacing an existing name, including races. + try { linkSync(temporary, path); } + catch (err) { + if ((err as NodeJS.ErrnoException).code === "EEXIST" && existingRole(path) === content) return { path, created: false }; + throw err; + } + } finally { unlinkSync(temporary); } + return { path, created: true }; +} diff --git a/plugins/codexclaw/components/subagent-config/test/role-registration.test.ts b/plugins/codexclaw/components/subagent-config/test/role-registration.test.ts new file mode 100644 index 00000000..a251ebc1 --- /dev/null +++ b/plugins/codexclaw/components/subagent-config/test/role-registration.test.ts @@ -0,0 +1,79 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdtempSync, mkdirSync, readFileSync, writeFileSync, rmSync, symlinkSync, existsSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { supportsSymlinks, symlinkDirSync } from "../../cxc-ops/test-support/symlink-support.ts"; +import { execFile } from "node:child_process"; +import { promisify } from "node:util"; +import { fileURLToPath } from "node:url"; +import { registerExecutor } from "../src/role-registration.ts"; +import { parseSubagentsArgs } from "../src/cli.ts"; + +test("register executor publishes complete role, omits model sentinel and preserves user settings", (t) => { + const home = mkdtempSync(join(tmpdir(), "executor-register-")); + t.after(() => rmSync(home, { recursive: true, force: true })); + writeFileSync(join(home, "config.toml"), "# user config\n"); + mkdirSync(join(home, "agents")); + writeFileSync(join(home, "agents/worker.toml"), "# legacy user role\n"); + const result = registerExecutor(home); + assert.equal(result.created, true); + const content = readFileSync(result.path, "utf8"); + const template = readFileSync(new URL("../../../agents/executor.toml", import.meta.url), "utf8"); + assert.equal(content.split('developer_instructions = ')[1], template.split('developer_instructions = ')[1]); + assert.match(content, /^name = "executor"$/m); + assert.doesNotMatch(content, /^(model|model_reasoning_effort|sandbox_mode|approval_policy)\s*=/m); + assert.deepEqual(registerExecutor(home), { path: result.path, created: false }); + assert.equal(readFileSync(join(home, "config.toml"), "utf8"), "# user config\n"); + assert.equal(readFileSync(join(home, "agents/worker.toml"), "utf8"), "# legacy user role\n"); +}); + +test("registration rejects conflicting file and directory without replacing them", (t) => { + for (const kind of ["file", "directory"]) { + const home = mkdtempSync(join(tmpdir(), "executor-conflict-")); + t.after(() => rmSync(home, { recursive: true, force: true })); + mkdirSync(join(home, "agents")); + const path = join(home, "agents/executor.toml"); + if (kind === "file") writeFileSync(path, "# custom"); else mkdirSync(path); + assert.throws(() => registerExecutor(home), /differs|non-regular/); + if (kind === "file") assert.equal(readFileSync(path, "utf8"), "# custom"); + } +}); + +for (const kind of ["directory", "role"] as const) { + test(`registration refuses symlink ${kind}`, (t) => { + const support = supportsSymlinks(); + if (kind === "directory" ? !support.dir : !support.file) { t.skip("host cannot create this symlink type"); return; } + const home = mkdtempSync(join(tmpdir(), "executor-symlink-")); + t.after(() => rmSync(home, { recursive: true, force: true })); + const outside = join(home, "outside"); mkdirSync(outside); + if (kind === "directory") symlinkDirSync(outside, join(home, "agents")); + else { mkdirSync(join(home, "agents")); symlinkSync(join(outside, "missing"), join(home, "agents/executor.toml")); } + assert.throws(() => registerExecutor(home), /non-regular/); + assert.equal(existsSync(join(outside, "executor.toml")), false); + assert.equal(existsSync(join(outside, "missing")), false); + }); +} + + +test("register parser accepts only executor with no extra arguments", () => { + assert.deepEqual(parseSubagentsArgs(["register", "executor"]), { action: "register", role: "executor" }); + for (const args of [["register"], ["register", "worker"], ["register", "executor", "--force"]]) { + assert.ok(parseSubagentsArgs(args).error); + } +}); + +test("native CLI registers concurrently in CODEX_HOME and refuses later conflicting content", async (t) => { + const home = mkdtempSync(join(tmpdir(), "executor-cli-")); + t.after(() => rmSync(home, { recursive: true, force: true })); + const cli = fileURLToPath(new URL("../src/cli.ts", import.meta.url)); + const run = () => promisify(execFile)(process.execPath, [cli, "subagents", "register", "executor"], { env: { ...process.env, CODEX_HOME: home } }); + const results = await Promise.all([run(), run()]); + assert.equal(results.filter(r => r.stdout.startsWith("Registered:")).length, 1); + assert.equal(results.filter(r => r.stdout.startsWith("Already registered:")).length, 1); + const role = join(home, "agents/executor.toml"); + assert.match(readFileSync(role, "utf8"), /^name = "executor"$/m); + writeFileSync(role, "# user's customized executor\n"); + await assert.rejects(run(), /Existing executor role differs/); + assert.equal(readFileSync(role, "utf8"), "# user's customized executor\n"); +}); diff --git a/plugins/codexclaw/test/hook-e2e.test.mjs b/plugins/codexclaw/test/hook-e2e.test.mjs index e4974183..fcaf4f7d 100644 --- a/plugins/codexclaw/test/hook-e2e.test.mjs +++ b/plugins/codexclaw/test/hook-e2e.test.mjs @@ -713,7 +713,7 @@ test("agbrowse: user-prompt-submit hook e2e - natural language agbrowse request // lazygap_impl 010: SubagentStop evidence-receipt gate. A gated worker child with no // receipt must be blocked (decision:block); a valid receipt under .codexclaw/evidence/ // releases. Drives the real dist entrypoint via the manifest command. -test("L010: subagent-stop hook e2e - worker w/o receipt blocks, valid receipt releases", () => { +for (const agentType of ["executor", "worker"]) test(`L010: subagent-stop hook e2e - ${agentType} w/o receipt blocks, valid receipt releases`, () => { const { event, hookEvent, distAbs } = readHookCommand("./hooks/subagent-stop-verifying-evidence.json"); assert.equal(event, "SubagentStop"); const ep = snapshotEntrypoint(distAbs); @@ -723,7 +723,7 @@ test("L010: subagent-stop hook e2e - worker w/o receipt blocks, valid receipt re // 1) worker, no receipt -> block with the EVIDENCE_RECORDED contract. const blocked = runHook(ep, hookEvent, { hook_event_name: "SubagentStop", session_id: "s1", cwd: tmp, - agent_type: "worker", agent_id: "a1", last_assistant_message: "all done!", + agent_type: agentType, agent_id: "a1", last_assistant_message: "all done!", }); assert.equal(blocked.status, 0, blocked.stderr); const out = JSON.parse(blocked.stdout); @@ -743,7 +743,7 @@ test("L010: subagent-stop hook e2e - worker w/o receipt blocks, valid receipt re writeFileSync(join(tmp, ".codexclaw", "evidence", "p.md"), "tests green"); const ok = runHook(ep, hookEvent, { hook_event_name: "SubagentStop", session_id: "s3", cwd: tmp, - agent_type: "worker", agent_id: "a3", + agent_type: agentType, agent_id: "a3", last_assistant_message: "done.\nEVIDENCE_RECORDED: .codexclaw/evidence/p.md", }); assert.equal(ok.status, 0, ok.stderr); From 541991f17d8cd09dc9cbb33b1ee03ad1beb2afea Mon Sep 17 00:00:00 2001 From: thisisjun786 <259586770+thisisjun786@users.noreply.github.com> Date: Tue, 8 Sep 2026 05:59:28 +0000 Subject: [PATCH 005/108] test: verify registration conflicts and publish measured suite total --- README.ko.md | 2 +- README.md | 2 +- README.zh.md | 2 +- .../subagent-config/test/role-registration.test.ts | 5 ++++- 4 files changed, 7 insertions(+), 4 deletions(-) diff --git a/README.ko.md b/README.ko.md index 7230faba..6f5283f9 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,7 +13,7 @@

CI - 2,670 tests passing + 2,681 tests passing 28 skills 23 hooks Documentation diff --git a/README.md b/README.md index 1c4c163d..52201094 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@

CI - 2,670 tests passing + 2,681 tests passing 28 skills 23 hooks Documentation diff --git a/README.zh.md b/README.zh.md index a995b8f2..de8602b3 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,7 +13,7 @@

CI - 2,670 tests passing + 2,681 tests passing 28 skills 23 hooks Documentation diff --git a/plugins/codexclaw/components/subagent-config/test/role-registration.test.ts b/plugins/codexclaw/components/subagent-config/test/role-registration.test.ts index a251ebc1..72b09edd 100644 --- a/plugins/codexclaw/components/subagent-config/test/role-registration.test.ts +++ b/plugins/codexclaw/components/subagent-config/test/role-registration.test.ts @@ -74,6 +74,9 @@ test("native CLI registers concurrently in CODEX_HOME and refuses later conflict const role = join(home, "agents/executor.toml"); assert.match(readFileSync(role, "utf8"), /^name = "executor"$/m); writeFileSync(role, "# user's customized executor\n"); - await assert.rejects(run(), /Existing executor role differs/); + await assert.rejects(run(), (err: NodeJS.ErrnoException & { stdout?: string }) => { + assert.match(err.stdout ?? "", /Existing executor role differs/); + return true; + }); assert.equal(readFileSync(role, "utf8"), "# user's customized executor\n"); }); From 5484d7013f6e4a9df8713b5ae62b954e46eb54ae Mon Sep 17 00:00:00 2001 From: thisisjun786 <259586770+thisisjun786@users.noreply.github.com> Date: Tue, 8 Sep 2026 06:00:28 +0000 Subject: [PATCH 006/108] docs: record local executor proof and hook activation requirement --- .../002_local_verification.md | 31 +++++++++++++++++++ 1 file changed, 31 insertions(+) create mode 100644 devlog/_plan/260908_executor_role_registration/002_local_verification.md diff --git a/devlog/_plan/260908_executor_role_registration/002_local_verification.md b/devlog/_plan/260908_executor_role_registration/002_local_verification.md new file mode 100644 index 00000000..f09abbe3 --- /dev/null +++ b/devlog/_plan/260908_executor_role_registration/002_local_verification.md @@ -0,0 +1,31 @@ +# Local verification and activation boundary + +## Verified candidate +- Full `TMPDIR=/var/tmp/cxc-executor-01a07d17 npm test`: 2681 total, 2611 passed, 0 failed, 70 existing conditional skips. +- `npm run build`: 161 files compiled, layout validated. `npm run gate`: OK. `npm run smoke`: platform smoke OK on linux. +- New registration CLI/module strict tsc: exit 0. Broader touched import graph: 10 diagnostics, all reproduced unchanged on untouched 6d70ef4; no typecheck-clean claim for the whole repository. +- Registration tests cover native command invocation in a temporary CODEX_HOME, concurrent publication, idempotence, user file/config preservation, malformed arguments, conflicting directory/file, symlink rejection (host capability gated). +- Shipped SubagentStop entrypoint: executor and legacy worker without receipt block; a valid receipt releases. This is an invoked-entrypoint test, not native hook delivery proof. + +## Local application +17 payload files applied after checking each old byte sequence against base 6d70ef4. Backup and SHA manifest: /home/jun/tmp/cxc-update-20260908-01a07d17/executor-backup/manifest.json. +`cxc subagents register executor` created /home/jun/.codex/agents/executor.toml; repeated invocation reported Already registered. Python TOML parse confirms name executor, no model or sandbox override. Existing worker.toml and project subagents.json untouched. Global AGENTS.md implementation role now executor; previous global file backed up. + +## Native probe +New Codex 0.153.4 exec session 01a07f97-60f6-70f1-a685-e963f93e3d62 spawned child 01a07f97-a468-7963-b658-d321e26e91a4 after checking exposed executor role. Child wrote exactly EXECUTOR_NATIVE_OK, parent read back and closed child. Child confirmed native Role: scoped executor instructions plus INLINE_EXECUTOR_PROBE. Trace: /home/jun/tmp/cxc-update-20260908-01a07d17/native-executor-smoke/run.jsonl. Ephemeral execution does not retain session_meta; effective role is recorded by the probe's tool use/report, not independently recovered from persistent metadata. + +## Pending user activation +Before patch: doctor overall PASS, 24 trusted hook hashes. After patch: +`[FAIL] hook-trust: drifted codexclaw@codexclaw:hooks/subagent-stop-verifying-evidence.json:subagent_stop:0:0 expected=sha256:84e1bb4945bc1f0c31d20ff1cfd3dc555eb266362cc159182962fc6c71f2e6d8 actual=sha256:9afd7aeccc4c240163001eec376823a6566ce30572fafcc300ceb1d6bb4c6290` +`overall: FAIL` +Normal hook reapproval and session restart are required. No trust records were edited. Actual executor SubagentStop delivery is UNVERIFIED until that approval; unit/dist tests do not substitute for it. Native probe did not change hook trust, role registration or FSM state. All task-owned subprocesses and probe child finished. + +## QA matrix +| Surface/scenario | Result | Evidence | +|---|---|---| +| CLI first register + repeat | PASS | installed CLI output and role TOML | +| CLI conflict + symlink + concurrent registration | PASS | role-registration.test.ts and suite log | +| v1/v2 executor model/effort vs review keywords | PASS | spawn-attach-hook.test.ts | +| canonical/legacy exit evidence | PASS (entrypoint) | hook-e2e.test.mjs | +| native executor dispatch + scoped file readback | PASS (probe report) | native trace and executor-proof.txt | +| native changed hook activation | PENDING user reapproval | doctor output above | From b23d5e17b3e53d361d655d1641e1d66cd49af1d7 Mon Sep 17 00:00:00 2001 From: thisisjun786 <259586770+thisisjun786@users.noreply.github.com> Date: Tue, 8 Sep 2026 06:02:47 +0000 Subject: [PATCH 007/108] docs: align remaining executor setup and QA guidance --- README.zh.md | 14 ++++++++++++-- .../002_local_verification.md | 2 ++ plugins/codexclaw/agents/README.md | 3 ++- plugins/codexclaw/skills/qa/SKILL.md | 4 ++-- 4 files changed, 18 insertions(+), 5 deletions(-) diff --git a/README.zh.md b/README.zh.md index de8602b3..c966067a 100644 --- a/README.zh.md +++ b/README.zh.md @@ -49,13 +49,23 @@ IDLE ── P ── A ── B ── C ── D ── IDLE ## 安装 -两行命令即可完成安装。无需构建、无需 npm install、无需修改配置文件。 +安装插件后,需注册一次实现角色。无需构建或 npm install。 ```bash codex plugin marketplace add https://github.com/lidge-jun/codexclaw codex plugin add codexclaw@codexclaw ``` +实现角色统一命名为 `executor`。使用安装目录中的 CLI 注册: + +```sh +node "/bin/cxc.mjs" subagents register executor +``` + +启动新的 Codex 会话,确认调用工具的角色列表包含 `executor`。 +注册命令保留现有用户角色和模型设置,不会覆盖同名的不同文件。 +参见[角色注册与旧 worker 兼容说明](plugins/codexclaw/agents/README.md)。 + 然后重启 Codex,并在弹出的审批中批准 22 个 hooks(升级后需再次批准——内容哈希信任模型)。既可以直接在聊天中使用,终端界面也随包提供——payload 自带 `cxc` 调度器,代理的 `cxc orchestrate` 命令在任何安装方式下都能运行: - `orchestrate status` — 查看 PABCD 状态机 @@ -111,7 +121,7 @@ plugins/codexclaw/ │ ├── pre-tool-use-* skill attach, goal guards, patch lint, interview guard │ ├── post-tool-use-* interview capture, render observation │ ├── stop-* PABCD continuation under active goals -│ ├── subagent-stop-* evidence verification for worker dispatches +│ ├── subagent-stop-* evidence verification for executor dispatches (legacy worker supported) │ └── post-compact-* cursor reinject, recall context, bg-terminal affordance │ ├── components/ 8 isolated feature modules (src + dist) diff --git a/devlog/_plan/260908_executor_role_registration/002_local_verification.md b/devlog/_plan/260908_executor_role_registration/002_local_verification.md index f09abbe3..f4071abd 100644 --- a/devlog/_plan/260908_executor_role_registration/002_local_verification.md +++ b/devlog/_plan/260908_executor_role_registration/002_local_verification.md @@ -29,3 +29,5 @@ Normal hook reapproval and session restart are required. No trust records were e | canonical/legacy exit evidence | PASS (entrypoint) | hook-e2e.test.mjs | | native executor dispatch + scoped file readback | PASS (probe report) | native trace and executor-proof.txt | | native changed hook activation | PENDING user reapproval | doctor output above | + +Final independent code review: 01a07f96-ca60-7e60-bae4-0b6dcbb4615e VERDICT: PASS, no blockers; reviewer independently ran219 focused tests +2 dist exit tests. Nonblocking README.zh and QA canonical-name guidance fixed; hard-link support documented. Same-user race EEXIST wording remains a nonblocking usability residual. diff --git a/plugins/codexclaw/agents/README.md b/plugins/codexclaw/agents/README.md index 57ca32c9..ec9381e9 100644 --- a/plugins/codexclaw/agents/README.md +++ b/plugins/codexclaw/agents/README.md @@ -28,7 +28,8 @@ This creates `$CODEX_HOME/agents/executor.toml` (default `~/.codex/agents/execut from the shipped executor prompt, omitting the plugin's `model = "default"` sentinel. The installed role does not override model, effort, sandbox or approval policy. Identical files are left unchanged; conflicting files and symlinks are refused without overwrite. -Existing worker files and project model settings are preserved. +Existing worker files and project model settings are preserved. Publication requires +filesystem hard-link support; unsupported filesystems fail without replacing a role. Start a new Codex session and check that the live spawn schema exposes `executor`. If it does not, or the host rejects that agent_type, report the unmet setup prerequisite; diff --git a/plugins/codexclaw/skills/qa/SKILL.md b/plugins/codexclaw/skills/qa/SKILL.md index 902cdaa0..40c5c0fc 100644 --- a/plugins/codexclaw/skills/qa/SKILL.md +++ b/plugins/codexclaw/skills/qa/SKILL.md @@ -175,8 +175,8 @@ A user-facing surface change closes C with BOTH: the automated gate (dev-testing) AND this skill's QA matrix — any FAIL verdict blocks the C>D claim until repaired (LOOP-REPAIR-01 counts apply) or the criterion is re-scoped through a P-phase amendment, never silently. This is E7 discipline: -no hook reads verdict.json. The E2 touchpoint: QA delegated to a `worker` -subagent rides the existing SubagentStop receipt gate — the worker cannot +no hook reads verdict.json. The E2 touchpoint: QA delegated to a registered `executor` +subagent rides the existing SubagentStop receipt gate (legacy `worker` also supported) — the executor cannot finish without a non-empty receipt under `.codexclaw/evidence/`. ## v2 candidates (deliberately not shipped) From 4b0bd1541224d136605a0e3d0d0b56ad2bfd0059 Mon Sep 17 00:00:00 2001 From: thisisjun786 <259586770+thisisjun786@users.noreply.github.com> Date: Tue, 8 Sep 2026 06:04:37 +0000 Subject: [PATCH 008/108] docs: archive executor workflow delivery evidence --- .../260908_executor_role_registration/000_plan.md | 3 +++ .../260908_executor_role_registration/001_evidence.md | 0 .../002_local_verification.md | 0 .../260908_executor_role_registration/010_executor.md | 0 4 files changed, 3 insertions(+) rename devlog/{_plan => _fin}/260908_executor_role_registration/000_plan.md (85%) rename devlog/{_plan => _fin}/260908_executor_role_registration/001_evidence.md (100%) rename devlog/{_plan => _fin}/260908_executor_role_registration/002_local_verification.md (100%) rename devlog/{_plan => _fin}/260908_executor_role_registration/010_executor.md (100%) diff --git a/devlog/_plan/260908_executor_role_registration/000_plan.md b/devlog/_fin/260908_executor_role_registration/000_plan.md similarity index 85% rename from devlog/_plan/260908_executor_role_registration/000_plan.md rename to devlog/_fin/260908_executor_role_registration/000_plan.md index d43b212c..dfacadde 100644 --- a/devlog/_plan/260908_executor_role_registration/000_plan.md +++ b/devlog/_fin/260908_executor_role_registration/000_plan.md @@ -26,3 +26,6 @@ Main owns role mapping, prompt/README/skill call guidance, model routing tests, ## Previous cycle Previous cycle fixed source worktree binding and ended IDLE. This distinct cycle fixes role identity; no old worktree/phase evidence is reused as proof. + +## Delivery conclusion +Implementation and local registration are complete; PR https://github.com/lidge-jun/codexclaw/pull/91 targets dev. Canonical executor naming, legacy compatibility and configuration-preserving registration passed independent review and local checks. Final local payload covers18files after documentation follow-up. No merge/release performed. Normal user hook reapproval is still required before claiming changed native SubagentStop activation; this is an explicit handoff prerequisite, not a passing hook-delivery claim. CI is tracked on the PR separately from local proof. CXC cycle closed to IDLE after verified local checks. diff --git a/devlog/_plan/260908_executor_role_registration/001_evidence.md b/devlog/_fin/260908_executor_role_registration/001_evidence.md similarity index 100% rename from devlog/_plan/260908_executor_role_registration/001_evidence.md rename to devlog/_fin/260908_executor_role_registration/001_evidence.md diff --git a/devlog/_plan/260908_executor_role_registration/002_local_verification.md b/devlog/_fin/260908_executor_role_registration/002_local_verification.md similarity index 100% rename from devlog/_plan/260908_executor_role_registration/002_local_verification.md rename to devlog/_fin/260908_executor_role_registration/002_local_verification.md diff --git a/devlog/_plan/260908_executor_role_registration/010_executor.md b/devlog/_fin/260908_executor_role_registration/010_executor.md similarity index 100% rename from devlog/_plan/260908_executor_role_registration/010_executor.md rename to devlog/_fin/260908_executor_role_registration/010_executor.md From a202ecf844cee5d976463b1f29da66d0ebc82517 Mon Sep 17 00:00:00 2001 From: Joonsuh Park Date: Tue, 8 Sep 2026 16:00:34 +0900 Subject: [PATCH 009/108] test: cover Windows short-path source bindings --- README.ko.md | 2 +- README.md | 2 +- README.zh.md | 2 +- .../030_short_path_regression.md | 32 ++++++++++++++++ .../test/worktree-source-integration.test.ts | 37 ++++++++++++++++++- 5 files changed, 71 insertions(+), 4 deletions(-) create mode 100644 devlog/_plan/260907_worktree_source_binding/030_short_path_regression.md diff --git a/README.ko.md b/README.ko.md index 0ecb4738..35c1a4ca 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,7 +13,7 @@

CI - 2,670 tests passing + 2,671 tests passing 28 skills 23 hooks Documentation diff --git a/README.md b/README.md index c567ba71..da98c87b 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@

CI - 2,670 tests passing + 2,671 tests passing 28 skills 23 hooks Documentation diff --git a/README.zh.md b/README.zh.md index a995b8f2..d339d416 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,7 +13,7 @@

CI - 2,670 tests passing + 2,671 tests passing 28 skills 23 hooks Documentation diff --git a/devlog/_plan/260907_worktree_source_binding/030_short_path_regression.md b/devlog/_plan/260907_worktree_source_binding/030_short_path_regression.md new file mode 100644 index 00000000..b177b12b --- /dev/null +++ b/devlog/_plan/260907_worktree_source_binding/030_short_path_regression.md @@ -0,0 +1,32 @@ +# Preserve Windows 8.3 regression coverage + +PR #84 merged the native-path fix in bb204ead04935a8970c3b9f70f9019eebb74e8f6. +The runtime behavior and fixture canonicalization are already present on dev and +main. The actual short-name regression from thisisjun786/codexclaw#1 was not +included, so this follow-up carries only that coverage forward. + +The test asks cmd.exe for the fixture's real 8.3 alias, binds using short paths, +then checks resolution and byte-preserving repeat binding using long paths. +It also checks the source command's native cwd, pinned source root, B/C +progression, and a validated receipt executed in the linked worktree. +The test explicitly skips outside Windows or when the temporary volume does +not provide 8.3 aliases. Production code is unchanged. + +Verification on Node 24.15.0: + +- Windows: the new regression passed without skipping. +- WSL Ubuntu: the affected integration file passed 18 tests with zero failures; + the Windows-only regression skipped. This includes the damaged-symlink case + and the shipped CLI flow. +- Repository gate and whitespace checks passed. +- Full Windows suite: 2,671 tests, 2,580 passed, 81 skipped, 10 failed. The + failures are the nine existing session-binding symlink fixtures and the + damaged-symlink integration case on this host without symlink creation + permission. The new regression passed. No failing test was weakened or + suppressed; this is not a green full-suite result. +- Published test-count badges were updated from the measured total using + `inventory.mjs --write --tests 2671`; the measured-count check passed. + +The original version of this regression was verified red before the native-path +fix and green after it in the earlier follow-up. The current run checks the +upstream implementation without replacing it with that earlier patch. diff --git a/plugins/codexclaw/components/pabcd-state/test/worktree-source-integration.test.ts b/plugins/codexclaw/components/pabcd-state/test/worktree-source-integration.test.ts index 77239311..20b6f6f0 100644 --- a/plugins/codexclaw/components/pabcd-state/test/worktree-source-integration.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/worktree-source-integration.test.ts @@ -3,10 +3,11 @@ import assert from "node:assert/strict"; import { mkdtempSync, mkdirSync, realpathSync, writeFileSync, readFileSync, rmSync, symlinkSync } from "node:fs"; import { fileURLToPath } from "node:url"; import { tmpdir } from "node:os"; -import { join } from "node:path"; +import { dirname, join } from "node:path"; import { execFileSync } from "node:child_process"; import { DatabaseSync } from "node:sqlite"; import { runSessionCli } from "../src/session-cli.ts"; +import { bindSessionSource, resolveSessionSource } from "../src/session-source.ts"; import { defaultState, writeState, readState } from "../src/state.ts"; import { runOrchestrateCli, parseOrchestrateCliArgs } from "../src/orchestrate-cli.ts"; import { runReceiptCli } from "../src/receipt-cli.ts"; @@ -46,6 +47,40 @@ function bind(f: ReturnType) { assert.equal(r.code, 0, r.output); } +test("Windows short paths bind and resolve the same immutable worktree", { skip: process.platform !== "win32" }, t => { + const f = fixture(t); + const root = dirname(f.cwd); + const shortRoot = execFileSync("cmd.exe", ["/d", "/c", 'for %I in ("%CXC_TEST_LONG_PATH%") do @echo %~sI'], { + env: { ...process.env, CXC_TEST_LONG_PATH: root }, encoding: "utf8", windowsVerbatimArguments: true, + }).trim(); + if (shortRoot === root) { t.skip("8.3 names are unavailable on the temporary volume"); return; } + assert.equal(realpathSync.native(shortRoot), root); + const shortCwd = join(shortRoot, "native"), shortSource = join(shortRoot, "source"); + assert.equal(bindSessionSource(shortCwd, id, shortSource), f.source); + const path = join(f.cwd, ".codexclaw", "sources", `${id}.json`); + const bytes = readFileSync(path, "utf8"); + const binding = JSON.parse(bytes); + assert.equal(binding.nativeCwd, f.cwd); + assert.equal(binding.sourceRoot, f.source); + assert.equal(resolveSessionSource(f.cwd, id), f.source); + assert.equal(resolveSessionSource(shortCwd, id), f.source); + assert.equal(bindSessionSource(f.cwd, id, f.source), f.source); + const bound = runSessionCli(["source", shortSource, "--json"], f.cwd, f.env); + assert.equal(bound.code, 0, bound.output); + assert.equal(JSON.parse(bound.output).cwd, f.cwd); + assert.equal(JSON.parse(bound.output).sourceCwd, f.source); + assert.equal(readFileSync(path, "utf8"), bytes); + assert.equal(edge(shortCwd, "B", "A").code, 0); + assert.equal(readState(f.cwd, id).boundSourceRoot, f.source); + writeFileSync(join(f.source, "implemented"), "yes"); + const check = edge(f.cwd, "C", "B"); + assert.equal(check.code, 0, check.output); + const receipt = runReceiptCli({ verb: "test", cwd: f.cwd, session: id, + command: [process.execPath, "-e", `require('node:assert/strict').equal(process.cwd(), ${JSON.stringify(f.source)})`] }); + assert.equal(receipt.code, 0, receipt.output); + assert.equal(validateCheckReceipt(readState(f.cwd, id), id, receipt.output, f.cwd).ok, true); +}); + test("worktree change advances B and Check runs there while receipts stay native", t => { const f = fixture(t); bind(f); assert.equal(edge(f.cwd, "B", "A").code, 0); From eb39a12ffbfdc5ede98817247ba1bde5cca23a8e Mon Sep 17 00:00:00 2001 From: thisisjun786 <259586770+thisisjun786@users.noreply.github.com> Date: Tue, 8 Sep 2026 09:55:13 +0000 Subject: [PATCH 010/108] fix(subagents): persist effort, add global defaults and refresh OCX models --- README.ko.md | 2 +- README.md | 10 +- README.zh.md | 2 +- .../000_plan.md | 16 ++ .../010_implementation.md | 29 +++ .../020_verification.md | 30 +++ .../_plan/260908_subagent_scope/000_plan.md | 23 ++ .../260908_subagent_scope/001_verification.md | 17 ++ docs/security-hardening.md | 11 +- package.json | 2 +- .../messenger-bridge/dist/api-compat.js | 51 +---- .../messenger-bridge/dist/gateway-commands.js | 9 +- .../dist/telegram-commands.js | 2 +- .../dist/telegram-interactive.js | 28 ++- .../messenger-bridge/dist/win-exec.js | 91 +------- .../messenger-bridge/src/api-compat.ts | 51 +---- .../messenger-bridge/src/gateway-commands.ts | 9 +- .../messenger-bridge/src/telegram-commands.ts | 2 +- .../src/telegram-interactive.ts | 28 ++- .../messenger-bridge/src/win-exec.ts | 91 +------- .../test/gateway-commands.test.ts | 3 +- .../test/subagent-effort.test.ts | 124 +++++++++++ .../test/telegram-interactive.test.ts | 11 +- .../subagent-config/dist/catalog.js | 109 +++++----- .../components/subagent-config/dist/cli.js | 31 ++- .../subagent-config/dist/live-catalog.js | 119 +++++++++++ .../components/subagent-config/dist/mcp.js | 49 ++--- .../subagent-config/dist/settings-api.js | 31 +++ .../components/subagent-config/dist/store.js | 193 +++++++++++------ .../subagent-config/dist/win-exec.js | 89 ++++++++ .../components/subagent-config/src/catalog.ts | 111 +++++----- .../components/subagent-config/src/cli.ts | 33 ++- .../subagent-config/src/live-catalog.ts | 119 +++++++++++ .../components/subagent-config/src/mcp.ts | 49 ++--- .../subagent-config/src/settings-api.ts | 31 +++ .../components/subagent-config/src/store.ts | 197 ++++++++++++------ .../subagent-config/src/win-exec.ts | 89 ++++++++ .../subagent-config/test/catalog.test.ts | 23 +- .../subagent-config/test/global-home.test.ts | 46 ++++ .../subagent-config/test/live-catalog.test.ts | 85 ++++++++ .../test/scoped-surfaces.test.ts | 74 +++++++ .../subagent-config/test/scopes.test.ts | 103 +++++++++ .../subagent-config/test/store.test.ts | 8 +- plugins/codexclaw/gui/src/App.tsx | 3 +- plugins/codexclaw/gui/src/api.ts | 61 ++++-- .../gui/src/components/EffortSelect.tsx | 7 +- .../gui/src/components/ModelSelect.tsx | 16 +- plugins/codexclaw/gui/src/pages/Dashboard.tsx | 16 +- plugins/codexclaw/gui/src/pages/Subagents.tsx | 153 ++++++++------ plugins/codexclaw/gui/src/server/handlers.ts | 56 +---- .../codexclaw/gui/src/server/middleware.ts | 11 +- plugins/codexclaw/gui/src/ui/help.tsx | 11 +- plugins/codexclaw/gui/test/handlers.test.ts | 20 +- .../gui/test/subagent-client.test.ts | 48 +++++ .../gui/test/subagent-scope-api.test.ts | 50 +++++ plugins/codexclaw/scripts/test.mjs | 14 ++ plugins/codexclaw/test/hook-e2e.test.mjs | 4 +- 57 files changed, 1910 insertions(+), 791 deletions(-) create mode 100644 devlog/_fin/260908_global_settings_catalog/000_plan.md create mode 100644 devlog/_fin/260908_global_settings_catalog/010_implementation.md create mode 100644 devlog/_fin/260908_global_settings_catalog/020_verification.md create mode 100644 devlog/_plan/260908_subagent_scope/000_plan.md create mode 100644 devlog/_plan/260908_subagent_scope/001_verification.md create mode 100644 plugins/codexclaw/components/messenger-bridge/test/subagent-effort.test.ts create mode 100644 plugins/codexclaw/components/subagent-config/dist/live-catalog.js create mode 100644 plugins/codexclaw/components/subagent-config/dist/settings-api.js create mode 100644 plugins/codexclaw/components/subagent-config/dist/win-exec.js create mode 100644 plugins/codexclaw/components/subagent-config/src/live-catalog.ts create mode 100644 plugins/codexclaw/components/subagent-config/src/settings-api.ts create mode 100644 plugins/codexclaw/components/subagent-config/src/win-exec.ts create mode 100644 plugins/codexclaw/components/subagent-config/test/global-home.test.ts create mode 100644 plugins/codexclaw/components/subagent-config/test/live-catalog.test.ts create mode 100644 plugins/codexclaw/components/subagent-config/test/scoped-surfaces.test.ts create mode 100644 plugins/codexclaw/components/subagent-config/test/scopes.test.ts create mode 100644 plugins/codexclaw/gui/test/subagent-client.test.ts create mode 100644 plugins/codexclaw/gui/test/subagent-scope-api.test.ts create mode 100644 plugins/codexclaw/scripts/test.mjs diff --git a/README.ko.md b/README.ko.md index 0ecb4738..7ff30f28 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,7 +13,7 @@

CI - 2,670 tests passing + 2,692 tests passing 28 skills 23 hooks Documentation diff --git a/README.md b/README.md index c567ba71..e24186f0 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@

CI - 2,670 tests passing + 2,692 tests passing 28 skills 23 hooks Documentation @@ -41,6 +41,14 @@ IDLE ── P ── A ── B ── C ── D ── IDLE **Multi-Model Subagents** — role-based dispatch (explorer / reviewer / executor) with per-role model and prompt overrides. Configuration persists across sessions and applies automatically through the spawn-wrapper hook. A local GUI (Vite + React) provides visual config and, when opencodex is detected, a provider link bar. (Dashboard: build from a repo checkout for now; bundled in a follow-up release.) +Subagent settings resolve **per role: project → global → original session**. Open **Global Settings** to edit user defaults in `$CODEXCLAW_HOME/subagents.json` (default `~/.codexclaw/subagents.json`). The existing **Subagents** page edits `/.codexclaw/subagents.json`: each model dropdown offers **Main model**, **Global settings**, and individual models. Global settings follows the entire role's defaults, including effort and prompt; choose a main/direct model to customize that project role. Existing explicit project entries and `effort: null` retain their meaning. Main model changes only the model source; session effort separately inherits the original session's effort. + +The shared catalog reads OCX's enabled models with the non-mutating `ocx models live --json`, refreshes after a short cache lifetime, and supports **Refresh models**. Disabled or pending models are excluded. If OCX is absent it reads the configured Codex catalog (`model_catalog_json`, with `CODEX_MODELS_CACHE_PATH` override). An unavailable source yields an explicit error or labeled last-known list, never a fabricated four-model roster. The dashboard restricts effort choices to the model's advertised supported values; CLI/MCP validation still validates wire values only, not model-specific compatibility. + +CLI list/get/set/reset accept trailing `--global`. MCP `subagents_get`/`subagents_set` and GET `/api/subagents?scope=global` / POST `scope: "global"` share the same store. `inherit: true` removes the selected scope's entire role override; `effort: null` only clears effort. The former unpublished `$CODEX_HOME/codexclaw/subagents.json` path is read only if the canonical file is absent and `CODEXCLAW_HOME` is not explicitly set. The first explicit global edit/reset preserves its other roles in the canonical file and leaves the old file untouched. + + + **Recall** — searches past Codex conversations and the memory store from disk artifacts before asking the user, so context survives session boundaries and compaction. **Repo Map** — `cxc map

` runs tree-sitter parsing + PageRank ranking to produce a structure overview of unfamiliar code, letting the agent orient before deep `rg` dives. (Repo checkout only — requires the vendored Python toolchain.) diff --git a/README.zh.md b/README.zh.md index a995b8f2..b9dcb4cc 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,7 +13,7 @@

CI - 2,670 tests passing + 2,692 tests passing 28 skills 23 hooks Documentation diff --git a/devlog/_fin/260908_global_settings_catalog/000_plan.md b/devlog/_fin/260908_global_settings_catalog/000_plan.md new file mode 100644 index 00000000..2818e555 --- /dev/null +++ b/devlog/_fin/260908_global_settings_catalog/000_plan.md @@ -0,0 +1,16 @@ +# Global settings and live model discovery + +Users should choose main/global/direct models in each existing role dropdown, manage user defaults on a separate Global Settings page, and see the models currently enabled in OCX. This replaces the Editing selector and ambiguous reset buttons from 81faa22 while retaining its effort persistence fix. + +- Archetype/trigger: satisfy-spec, Jun's explicit PABCD implementation request after UX review. +- Goal: separate global settings, clear role inheritance, canonical CXC home, truthful shared model list, tested in this host before any PR. +- Non-goals: PR/push, merge/release, installed plugin replacement, paid inference, changing existing user role preferences or unrelated catalog-native-models worktree. +- Verifiers: component tests, GUI strict tsc/build, actual serve HTTP restart fixtures, browser flows, read-only OCX roster comparison. Existing suite and build were verified in prior cycle (2,614 passes, 70 conditional skips); new commands directly target the changed files. Observe errors by disabling fixture server/returning malformed data; observe refresh by changing fake OCX output and forcing refresh. +- Stop: clean local commit, independent review and actual-host preview checks complete; report before PR. +- Artifacts: this unit's numbered plan/design/check docs and /home/jun/tmp/cxc-global-settings-01a07d17 runtime logs. +- Outcomes: implemented/tested local change, or explicit unresolved evidence; no automatic publication. +- Escalation: main handles routine implementation choices; ask only if required live mutation exceeds scope. Main reclaims stalled executor scope explicitly; no concurrent overlapping writes. + +Prior D: effort save + role-level project > global > session passed. Change direction because user rejected Editing/reset-button UX and requested CXC-owned global path plus automatic OCX discovery. Preserve entire-role inheritance semantics: absent project role follows global; selecting Global uses existing inherit:true reset; selecting a main/direct model creates/updates a project role and retains its effort/prompt. In global mode effort/prompt display inherited values read-only; pick main/direct model to customize. Existing persisted models and null effort are unchanged until an explicit edit. + +Design read: existing light dense developer dashboard, existing CSS tokens/type/icons, sidebar Global Settings entry. No new framework, assets, or expressive redesign. Original Subagents three-role layout stays. Each model selector offers Main model / Global settings / catalog entries. Global page offers Main model / catalog entries and effort/prompt controls. No Editing selector or per-role Use buttons. Status text shows actual inherited value and no ambiguity about write scope. Preserve disabled untrusted-project edits and prompt drafts. Existing Dashboard quick controls must use the same selection semantics. diff --git a/devlog/_fin/260908_global_settings_catalog/010_implementation.md b/devlog/_fin/260908_global_settings_catalog/010_implementation.md new file mode 100644 index 00000000..ebd50e88 --- /dev/null +++ b/devlog/_fin/260908_global_settings_catalog/010_implementation.md @@ -0,0 +1,29 @@ +# One implementation work-phase (C3, global config and subprocess boundary care) + +## Contracts and file map +- MODIFY subagent-config/src/store.ts: globalStorePath = CODEXCLAW_HOME/subagents.json or ~/.codexclaw/subagents.json. Export cxcHome(env). Read old CODEX_HOME/codexclaw/subagents.json only when new path absent AND CODEXCLAW_HOME is not explicitly set; first explicit global write copies raw legacy roles into canonical path, never deletes old file. Add tests proving isolated override, canonical precedence, untouched legacy and unrelated roles. No new RoleMode enum is needed; existing API inherit:true removes project override. +- MODIFY subagent-config/src/catalog.ts: native reader honors explicit cache path / configured root model_catalog_json / models_cache.json without fixed native-ID allowlist. Remove fabricated fallback entries (missing catalog => unavailable/error metadata). Keep buildCatalog for pure/native consumers and tests. Existing NATIVE_OPENAI_MODELS export may remain for compatibility but is never a fallback source. +- NEW subagent-config/src/live-catalog.ts: async readCatalog({forceRefresh?, env?, ...injectable deps}). Use read-only execFile('ocx',['models','live','--json']) with bounded timeout/output, no shell and no ensure/sync. Rows use namespaced ID, drop disabled or initialSelectionPending rows, preserve valid bare IDs. OCX absent => native file reader. Persist shared last-success catalog below CXC home, cache 30 seconds, coalesce concurrent requests, force refresh supports UI; successful empty roster is authoritative. On query failure use last-success as explicitly stale, otherwise unavailable. No raw stderr/credentials in response. Export metadata status fresh/stale/unavailable, source ocx/native, fetchedAt, message plus existing entries/state compatibility. Dynamic discovery/refresh does not write Codex or OCX preferences. +- MODIFY bridge api-compat.ts + GUI handlers/middleware + MCP: async catalog reader; GET /api/catalog?refresh=1 requests fresh roster, normal gets use cache. Preserve HTTP local guard. MCP catalog_list uses same reader; no duplicate sync implementation. +- MODIFY GUI api.ts: catalog response types, honest empty/error result, optional refresh. Existing setSubagentRole/getSettings APIs retained. +- MODIFY GUI Subagents.tsx, ModelSelect.tsx, Dashboard.tsx, App.tsx/help; NEW GlobalSettings.tsx or reuse parameterized role panel. Fixed scope by route, main/global/direct selection; global sentinel must not be stored as model ID. Global selection calls inherit:true; main/direct call mode/model patches. Disable inherited effort/prompt until customized, retain project trust warning guards. Sidebar Global Settings. Refresh on mount/focus and explicit refresh (bounded interval only if needed); display freshness/error and retry. Retain saved but unavailable model as labeled saved option, not an enabled catalog entry. +- UPDATE related tests, README/security/help; rebuild tracked component dist. Docs explain CODEXCLAW_HOME and CLI/MCP scope, no false four-model defaults. + +## Work allocation +Main owns store/global-path migration, API/MCP integration, source UI integration and acceptance. Executor lane owns ONLY catalog.ts + new live-catalog.ts + their tests (no shared dist/build). Its contract is above; main can work on disjoint store/UI. Independent reviewer audits plan and final result. Live role schema exposes registered executor; use it, no model override (configured routing applies). + +## Acceptance +1. Explicit main/global/direct dropdown transitions persist on GET and server restart; global mutation propagates to inheriting roles in another project, explicit project models/null remain. +2. No Editing dropdown / Use buttons; dedicated global page works on desktop/mobile; prompt draft and failed save preserve current input; inherited fields visually disabled; trusted and ignored project cases honest. +3. Native arbitrary IDs/configured catalog path and OCX disabled/pending filtering; live changes visible after refresh; 0 enabled models remains empty; timeout/nonzero/malformed payload => stale/unavailable without fabricated defaults. Cache shared across project cwd and survives service restart; explicit CXC home never leaks real state. +4. Real local OCX roster comparison through patched API, with global/project persistence tested in isolated CXC_HOME and project cwd. Native CODEX_HOME may remain real for read-only catalog discovery. Browser uses the built GUI and that verified backend. +5. Existing effort regression checks + affected suites + full suite/gate and strict typechecks; preserve unrelated work and runtime settings. Leave a task-owned preview running for user review; do not replace live installed service. + +## Audit fixes +- All changed/global-store tests set CODEXCLAW_HOME explicitly, including subagent-config scopes/scoped-surfaces/store fixtures, messenger subagent-effort child env, GUI subagent-scope-api, and suite runner. Audit each CODEX_HOME-only hook fixture; default no-config readers must also be isolated at suite entry. Canonical home deliberately does not follow CODEX_HOME; tests assert explicit CODEXCLAW_HOME wins, not the incorrect expectation that CODEX_HOME alone relocates CXC settings. +- Shared global raw reader selects legacy data before BOTH setRole and resetRole; reset writes canonical remaining roles even when the role existed only in legacy. Test reset-all canonical empty prevents legacy resurrection. +- gateway-commands.ts and telegram-interactive.ts consumers also move to async readCatalog. Main rebuilds all dist after executor finishes. Catalog-only CLI/MCP/HTTP/messenger consumers agree. +- CatalogEntry carries reasoningEfforts: string[]|null from OCX/native metadata. Null or [] only permits inherited effort in the UI, known lists restrict existing supported EFFORTS options. Preserve an existing unsupported value visibly with a warning; do not silently change it. A direct model change incompatible with current effort is blocked with a message to select session effort first (main-model choice remains available to leave global inheritance). This work does not expand the store's wire-effort enum; broader effort values remain a separate compatibility decision. +- Async subprocess uses existing win-exec resolver on Windows. Cache disk fetchedAt gates queries across processes and uses exclusive temp + atomic rename; cache payload validated on read. Never merge disabled native rows back into an authoritative live OCX list. Saved out-of-roster selections remain labeled separately. +- Follow-on audit: Telegram model callbacks must use a stable hash of the model ID, reject expired/ambiguous tokens, and never resolve a changing catalog by index. Existing index buttons expire safely. Main owns this compatibility test. +- Move the existing pure win-exec helper implementation to subagent-config/src/win-exec.ts and keep messenger-bridge/src/win-exec.ts as a compatibility re-export. Executor owns the new helper copy, main owns the re-export; this avoids a package back-edge while preserving all tested platform escaping. diff --git a/devlog/_fin/260908_global_settings_catalog/020_verification.md b/devlog/_fin/260908_global_settings_catalog/020_verification.md new file mode 100644 index 00000000..5ded62ea --- /dev/null +++ b/devlog/_fin/260908_global_settings_catalog/020_verification.md @@ -0,0 +1,30 @@ +# Local verification — 2026-09-08 + +Implementation complete; PR, push, installation and release are outside this delivery. + +| Contract | Evidence | Result | +| --- | --- | --- | +| Effort persistence and validation | Existing real serve regression suite; browser POST/GET/restart, null and invalid request without file mutation | Pass | +| Global home and migration | global-home tests: explicit CXC home, legacy read, set/reset migration, unknown fields and other roles preserved | Pass | +| Project precedence and independence | scopes and scoped-surfaces tests: two project paths, global update, explicit null, CLI/MCP persistence, trust guard | Pass | +| Live catalog | Patched serve API compared with read-only `ocx models live --json`: all 19 enabled IDs match, disabled/pending absent | Pass | +| Conditional catalog paths | live-catalog tests: TTL, force refresh, coalescing, cross-process reuse, empty roster, malformed/error/stale, native configured path, actual 12-second timeout and output limit | Pass | +| User interface | Built GUI on actual serve: separate Global Settings route, main/global/direct dropdown, inherited controls disabled, refresh/stale display, rejected save and failed load/retry, model effort restriction, ignored project controls | Pass | +| Rendering | Personally inspected 1280x960 global and 390x844 subagent screenshots; no horizontal overflow or page errors | Pass | +| Packaging | Rebuilt tracked runtime; force-tracked new live-catalog and win-exec dist modules; inventory gate | Pass | +| Automated checks | Full suite: 2692 tests, 2622 pass, 70 conditional skips, 0 failures; GUI and changed core strict TypeScript; component and GUI builds | Pass | +| Operator settings | SHA256/existence comparison of Lina role settings, both global preference locations and Codex config | Unchanged | + +Artifacts: `/home/jun/tmp/cxc-global-settings-01a07d17/` contains full-tests.log, browser.log, browser-smoke.mjs, build.log, gui-build.log, gate.log, tsc-gui.log, tsc-core.log and screenshots. Test child servers and disposable browser projects were stopped/removed in finally blocks. The suite runner now creates an isolated CXC home, preventing catalog cache writes to the operator home during npm test. + +Independent review found three blockers: ignored new dist modules, two obsolete allowlist assertions, and live catalog access in formatting tests. All fixed; formatting tests now inject the roster, and npm test isolates CXC state. Final independent verdict PASS: reviewer reran 54 neighboring tests in isolation, all passed, and verified the operator cache timestamp remained unchanged. Review-only unisolated probing had created `~/.codexclaw/model-catalog.json` (derived catalog cache, no preferences); it is not deleted because other sessions may use it. + +Model capability restrictions are UI-only; CLI/MCP retain the existing wire effort validation. Provider inference and next subagent-turn routing were not exercised. Shared catalog cache identity includes PATH; differing service environments may refresh separately, and a discovery outage retries on the next poll. These optimization limits do not alter saved choices. + +Task-owned preview: http://127.0.0.1:37235/#/settings, execution handle 12300; project `/home/jun/tmp/cxc-global-settings-01a07d17/preview/project`, CXC home sibling `preview/cxc`. It uses a copy of Lina's role preferences and reads the real OCX catalog. Original installed plugin and prior preview were not replaced. + +Workflow source evidence is explicit local paths. The native session cwd `/home/jun/tmp` is not a Git repository; the FSM has no bound source identity or goalplan test receipt. No such proof is claimed. + +## Pre-PR recheck + +User authorized PR publication after another complete check. Fresh full suite again measured 2692 total / 2622 pass / 70 conditional skips / 0 failures. Rebuilt component and GUI bundles; GUI and changed-core strict typechecks, Linux platform smoke and repeated real serve/browser acceptance passed. CI also checks the measured suite count: corrected the three README badges from 2670 to 2692 with the repository inventory generator, then verified `inventory.mjs --check --tests 2692` and gate. No behavior changes were needed for these checks. Upstream contribution target is `dev` (6d70ef4), confirmed by remote refs and the repository target-enforcement workflow. Evidence is in the existing artifact directory's `pr-check/` folder. diff --git a/devlog/_plan/260908_subagent_scope/000_plan.md b/devlog/_plan/260908_subagent_scope/000_plan.md new file mode 100644 index 00000000..04148295 --- /dev/null +++ b/devlog/_plan/260908_subagent_scope/000_plan.md @@ -0,0 +1,23 @@ +# Subagent effort persistence and scoped defaults (C3) + +## Outcome and compatibility +Fix cxc serve dropping effort. Add role-level project > global > session inheritance. +Global file: $CODEX_HOME/codexclaw/subagents.json (default ~/.codex/codexclaw/subagents.json). +A present project role keeps its existing entire configuration, including null effort meaning parent-session effort. Removing a role via inherit:true returns it to the next scope. No migration or changes to real user settings. + +## Diff contract +- store.ts: optional scope (project default), sparse persisted roles, effective readConfig, separate readSettings returning additive scope/sources metadata. setRole accepts scope, resetRole removes scoped role. Preserve unrelated raw data. Global resolution is also used when tracked project config fails its existing trust-token check. +- Shared settings API helper validates role, scope, reset and effort before any write. Bridge and Vite handlers call the same helper. GET scope query and POST scope field default to project; responses include roles, scope, sources. Existing request shapes remain accepted. +- CLI get/set support --global; reset supports role and --global. MCP get/set scope and inherit boolean mirror the API. +- GUI scope selector, actual value source, explicit role reset, session inheritance labels. Prevent save/load races. Preserve current visual language. +- Generated dist rebuilt through existing build script. Add user documentation at existing subagent docs location. + +## Work split +Main owns store, shared API, bridge/Vite handlers and persistence/trust tests. Worker owns GUI page/client and UI tests. CLI/MCP follow core store contract. Reviewer audits plan and final diff independently. No overlapping write sets. + +## Verification +First demonstrate existing effort failure (red.log). Child-process HTTP tests cover all efforts, omitted field, null, bad values rejecting atomically, persisted reads after process restart and preserved models. Add global fallback, project precedence, reset, global independence, invalid scope, and tracked-project trust coverage using isolated CODEX_HOME. GUI build/typecheck and behavior tests, existing component suites, build and repository gate. Runtime spawn resolution tested without paid inference. Actual Lina settings untouched. No deployment or process takeover. + +## Audit disposition +Reviewer PASS conditional on sparse compatibility tests, common trust resolution, and one global path helper. Accepted: legacy three-default-role shadowing test, sparse-role fallback test, shared effective readSettings/spawn resolver with trustWarning and source metadata, hook payload regression. Add `overrides` booleans so the UI can remove a present but untrusted project role. Edits merge the saved scoped role when present, preserving its model even if runtime ignores it. +Rebuttal: rejecting nonexistent CODEX_HOME is unnecessary and prevents first-use setup. The host-controlled global root is created on explicit global write, matching current project-store behavior; browser input cannot choose arbitrary paths. Test automatic creation inside an isolated environment. HTTP global mutation uses existing loopback/Host/JSON/local-header checks (C4 boundary care within C3 feature). diff --git a/devlog/_plan/260908_subagent_scope/001_verification.md b/devlog/_plan/260908_subagent_scope/001_verification.md new file mode 100644 index 00000000..b1d7cb73 --- /dev/null +++ b/devlog/_plan/260908_subagent_scope/001_verification.md @@ -0,0 +1,17 @@ +# Implementation and verification + +The serve route and Vite/MCP now share settings-api.ts, including effort validation and persistence. store.ts resolves whole roles project > global > session, retains explicit null/default entries, writes only the edited role, preserves unrelated JSON, and exposes effective sources plus project-trust warnings. CLI supports a trailing --global and role reset. The GUI shows scope/source/reset, rejects failed loads, serializes saves, and saves prompt drafts explicitly to avoid overlapping writes. + +## Evidence (2026-09-08, Linux / Node 24.20.0) + +- Before implementation, subagent-effort.test.ts failed: saving low returned null. The initial log is /home/jun/tmp/cxc-serve-effort-01a07d17/red.log. +- Final full suite: 2,684 tests; 2,614 passed, 70 existing conditional skips, zero failures. Command: TMPDIR=/var/tmp/cxc-effort-check-01a07d17 CODEX_HOME=/var/tmp/cxc-effort-check-01a07d17/codex-home npm test. The isolated TMPDIR avoids an unrelated /tmp/.git affecting root-discovery fixtures. +- Core strict TypeScript and GUI tsc passed; component and GUI builds passed. New dist/settings-api.js is tracked for clone/marketplace parity; packaging/freshness checks passed. +- HTTP child-process regressions cover low/medium/high/xhigh/null, omitted effort, malformed effort/scope/reset rejection without writes, model preservation, and new server processes reading the same files. Separate fixtures cover global/project/reset precedence and legacy all-default roles. +- Compiled CLI/MCP roundtrips and actual spawn-hook input/output passed, including global model/effort injection on both payload surfaces and full-history fork exclusions. Vite query and trust metadata match spawn resolution; global CSRF is rejected. +- Chromium browser smoke against the built GUI and real serve CLI passed: effort save/reload/restart; model preservation; global/project/reset/null; failed save retaining state; prompt persistence; load failure/retry; desktop 1280px and mobile 390px without horizontal overflow or page errors. Screenshots and script: /home/jun/tmp/cxc-serve-effort-01a07d17/. +- Independent reviewer: core PASS and final GUI PASS. Final fixes preserve the Dashboard’s non-throwing client and disable ignored project edits on both GUI surfaces while retaining reset. After these fixes, GUI strict tsc, build, 27 GUI tests, and browser checks for ignored settings and Dashboard load failure passed. Repository gate and diff whitespace check passed. + +## Boundaries + +All preference changes were isolated fixtures. No model or effort was selected for Lina or user-global settings. No installed plugin/service was modified or restarted; no inference, push, PR, merge or deployment was performed. Changes are recorded in a local commit only. The implementation is on fix/serve-subagent-effort in /home/jun/code-worktrees/codexclaw/serve-subagent-effort. Other worktrees remain untouched. diff --git a/docs/security-hardening.md b/docs/security-hardening.md index edec2601..1ccf19fd 100644 --- a/docs/security-hardening.md +++ b/docs/security-hardening.md @@ -18,7 +18,16 @@ PABCD state, recall, subagent configuration, and messenger bridge. it, spawn-time model, effort, and prompt overrides are ignored unless the operator reviews it, runs `cxc subagents trust-token`, and exports the printed value. The value binds the canonical repository path and exact config digest, - so it cannot silently trust another checkout or later edits. + so it cannot silently trust another checkout or later edits. Ignored project + overrides fall back to operator-owned global roles, then the original session. + The settings API and spawn hook share this resolver; the dashboard reports the + warning and effective source. Global subagent defaults are stored separately at + `$CODEXCLAW_HOME/subagents.json` (default `~/.codexclaw/subagents.json`). + Live model discovery invokes only the read-only OCX model command with bounded + execution/output and keeps its shared cache under CXC home. It does not write + Codex or OCX preferences. + Global writes use the same local HTTP guards and cannot select a request-supplied + filesystem path. Neither scope writes native Codex `config.toml` or agent files. - Recursive delegation is authorized with a parent-minted, project/session-bound, single-use capability. Public marker text is only a request from a root dispatcher; it is never authority when supplied by a child. Worker evidence diff --git a/package.json b/package.json index 34b622db..16cadd5c 100644 --- a/package.json +++ b/package.json @@ -21,7 +21,7 @@ "scripts": { "build": "node plugins/codexclaw/scripts/build.mjs", "gate": "node plugins/codexclaw/scripts/gate.mjs", - "test": "node --test --test-concurrency=1 \"plugins/codexclaw/components/pabcd-state/test/*.test.ts\" \"plugins/codexclaw/components/config-guard/test/*.test.ts\" \"plugins/codexclaw/components/cxc-ops/test/*.test.ts\" \"plugins/codexclaw/components/recall/test/*.test.ts\" \"plugins/codexclaw/components/provider-bridge/test/*.test.ts\" \"plugins/codexclaw/components/subagent-config/test/*.test.ts\" \"plugins/codexclaw/components/messenger-bridge/test/*.test.ts\" \"plugins/codexclaw/components/skill-search/test/*.test.ts\" \"plugins/codexclaw/gui/test/*.test.ts\" \"plugins/codexclaw/test/*.test.mjs\"", + "test": "node plugins/codexclaw/scripts/test.mjs \"plugins/codexclaw/components/pabcd-state/test/*.test.ts\" \"plugins/codexclaw/components/config-guard/test/*.test.ts\" \"plugins/codexclaw/components/cxc-ops/test/*.test.ts\" \"plugins/codexclaw/components/recall/test/*.test.ts\" \"plugins/codexclaw/components/provider-bridge/test/*.test.ts\" \"plugins/codexclaw/components/subagent-config/test/*.test.ts\" \"plugins/codexclaw/components/messenger-bridge/test/*.test.ts\" \"plugins/codexclaw/components/skill-search/test/*.test.ts\" \"plugins/codexclaw/gui/test/*.test.ts\" \"plugins/codexclaw/test/*.test.mjs\"", "smoke": "node plugins/codexclaw/scripts/platform-smoke.mjs" }, "overrides": { diff --git a/plugins/codexclaw/components/messenger-bridge/dist/api-compat.js b/plugins/codexclaw/components/messenger-bridge/dist/api-compat.js index 5ecf2b14..48eb76da 100644 --- a/plugins/codexclaw/components/messenger-bridge/dist/api-compat.js +++ b/plugins/codexclaw/components/messenger-bridge/dist/api-compat.js @@ -6,16 +6,16 @@ * the Vite dev middleware. When cxc serve hosts the built GUI statically those * routes must exist or role saves silently fail (A-audit finding 2). * - * Source of truth for the route semantics: gui/src/server/middleware.ts + - * gui/src/server/handlers.ts. This module mirrors them over the already + * Subagent settings semantics are shared with the Vite handlers through + * subagent-config/settings-api. Other routes mirror them over the already * COMPILED component dists (relative .js specifiers survive the build's * .ts→.js rewrite untouched and resolve identically from src/ and dist/). * Phase 6 unifies the GUI dev middleware onto this module. */ import { spawnSync } from "node:child_process"; // Compiled component dists — runtime-typed, so minimal local shapes below. -import { readConfig, setRole, ROLES } from "../../subagent-config/dist/store.js"; -import { buildCatalog } from "../../subagent-config/dist/catalog.js"; +import { getSettings, updateSettings, settingsResponse } from "../../subagent-config/dist/settings-api.js"; +import { readCatalog } from "../../subagent-config/dist/live-catalog.js"; import { detectOcx } from "../../provider-bridge/dist/detect.js"; import { splitLines } from "./text-lines.js"; @@ -51,41 +51,8 @@ function detectDeps() { }; } -/** Map provider detection to the catalog's provider input — mirrored from gui/src/server/handlers.ts. */ -function providerToCatalogInput(status ) { - if (status.mode === "provider") { - // ocx-synced models surface via the native config cache, not a live call. - return { mode: "provider", ocxModels: undefined }; - } - return { mode: status.mode === "error" ? "error" : "native" }; -} - -function getSubagentsRoute(cwd ) { - return { status: 200, body: readConfig(cwd) }; -} - -function postSubagentsRoute(cwd , body ) { - if (!body || typeof body !== "object") return { status: 400, body: { error: "missing body" } }; - const b = body ; - const role = b.role ; - if (!ROLES.includes(role)) { - return { status: 400, body: { error: `unknown role "${String(b.role)}"` } }; - } - const patch = {}; - if (b.mode !== undefined) patch.mode = b.mode; - if (b.model !== undefined) patch.model = b.model; - if (b.promptOverride !== undefined) patch.promptOverride = b.promptOverride; - try { - return { status: 200, body: setRole(cwd, role, patch) }; - } catch (err) { - return { status: 400, body: { error: err instanceof Error ? err.message : String(err) } }; - } -} - -function getCatalogRoute() { - const status = detectOcx(detectDeps()) ; - const catalog = buildCatalog({ providerStatus: providerToCatalogInput(status) }); - return { status: 200, body: catalog }; +async function getCatalogRoute(forceRefresh = false) { + return { status: 200, body: await readCatalog({ forceRefresh }) }; } function getProviderRoute() { @@ -100,14 +67,14 @@ export function apiCompatRoutes() { { method: "GET", path: "/api/subagents", - handler: (ctx) => getSubagentsRoute(ctx.cwd), + handler: (ctx, _body, url) => settingsResponse(() => getSettings(ctx.cwd, url.searchParams.get("scope") ?? undefined)), }, { method: "POST", path: "/api/subagents", - handler: (ctx, body) => postSubagentsRoute(ctx.cwd, body), + handler: (ctx, body) => settingsResponse(() => updateSettings(ctx.cwd, body)), }, - { method: "GET", path: "/api/catalog", handler: () => getCatalogRoute() }, + { method: "GET", path: "/api/catalog", handler: (_ctx, _body, url) => getCatalogRoute(url.searchParams.get("refresh") === "1") }, { method: "GET", path: "/api/provider", handler: () => getProviderRoute() }, ]; } diff --git a/plugins/codexclaw/components/messenger-bridge/dist/gateway-commands.js b/plugins/codexclaw/components/messenger-bridge/dist/gateway-commands.js index 5a6e67e2..07b17d53 100644 --- a/plugins/codexclaw/components/messenger-bridge/dist/gateway-commands.js +++ b/plugins/codexclaw/components/messenger-bridge/dist/gateway-commands.js @@ -8,7 +8,7 @@ import { realpathSync, statSync } from "node:fs"; import { homedir } from "node:os"; import { resolve } from "node:path"; -import { buildCatalog } from "../../subagent-config/dist/catalog.js"; +import { readCatalog } from "../../subagent-config/dist/live-catalog.js"; import { chunkEmbedDescription, } from "./discord-api.js"; @@ -45,6 +45,7 @@ import { chunkTelegramMessage } from "./telegram-format.js"; + export const GATEWAY_COMMANDS = [ @@ -315,7 +316,7 @@ async function handleModel(ctx ) if (arg) { // Reserved /model subcommands are checked before save-verbatim so model ids // named "list" or "reset" cannot be stored accidentally. - if (arg === "list") return modelListResult(); + if (arg === "list") return modelListResult(ctx.readModelCatalog); if (arg === "reset") { ctx.db.setBindingModel(binding.id, "default"); const next = effectiveModel(ctx.db.getBinding(binding.id) ?? binding, agent); @@ -436,8 +437,8 @@ export function validateWorkdir(input ) { } } -function modelListResult() { - const catalog = buildCatalog() ; +async function modelListResult(reader = readCatalog) { + const catalog = await reader() ; const groups = groupCatalogEntries(catalog.entries ?? []); const text = groups.length === 0 ? "No models found." diff --git a/plugins/codexclaw/components/messenger-bridge/dist/telegram-commands.js b/plugins/codexclaw/components/messenger-bridge/dist/telegram-commands.js index 9b7d5d56..e86c9f67 100644 --- a/plugins/codexclaw/components/messenger-bridge/dist/telegram-commands.js +++ b/plugins/codexclaw/components/messenger-bridge/dist/telegram-commands.js @@ -162,7 +162,7 @@ async function handleModel(ctx ) { const current = String(result?.data?.model ?? "default"); return { text: result?.text ?? `Current model: ${current}`, - keyboard: buildModelPicker(loadModelCatalog(), current, binding.id), + keyboard: buildModelPicker(await loadModelCatalog(), current, binding.id), }; } diff --git a/plugins/codexclaw/components/messenger-bridge/dist/telegram-interactive.js b/plugins/codexclaw/components/messenger-bridge/dist/telegram-interactive.js index 2b0c0cac..1c808ab2 100644 --- a/plugins/codexclaw/components/messenger-bridge/dist/telegram-interactive.js +++ b/plugins/codexclaw/components/messenger-bridge/dist/telegram-interactive.js @@ -2,9 +2,10 @@ * telegram-interactive.ts — inline keyboard callback encoding and dispatch. * * Telegram callback_data is capped at 64 bytes, so payloads stay compact and - * model selections use catalog indexes instead of full model ids. + * model selections use stable ID hashes instead of mutable catalog indexes. */ -import { buildCatalog } from "../../subagent-config/dist/catalog.js"; +import { readCatalog } from "../../subagent-config/dist/live-catalog.js"; +import { createHash } from "node:crypto"; import { AGENT_EFFORTS, AGENT_THREAD_MODES, AGENT_TOOL_PROGRESS_MODES, } from "./db.js"; import { telegramReplyThreadId, telegramTopicId, } from "./telegram-api.js"; @@ -61,8 +62,8 @@ export function decodeCallback(data ) { return { type, payload: rest.join(":") }; } -export function loadModelCatalog() { - const catalog = buildCatalog() ; +export async function loadModelCatalog() { + const catalog = await readCatalog() ; const entries = Array.isArray(catalog.entries) ? catalog.entries : []; return entries .filter((entry) => typeof entry.id === "string" && entry.id.length > 0) @@ -72,11 +73,15 @@ export function loadModelCatalog() { })); } +export function modelToken(id ) { + return "h" + createHash("sha256").update(id).digest("hex").slice(0, 24); +} + export function buildModelPicker(catalog , current , bindingId = 0) { return rows( - catalog.map((entry, index) => ({ + catalog.map((entry) => ({ text: `${entry.id === current ? "* " : ""}${entry.label ?? entry.id}`, - callback_data: encodeCallback({ type: "model_select", payload: `${bindingId}:${index}` }), + callback_data: encodeCallback({ type: "model_select", payload: `${bindingId}:${modelToken(entry.id)}` }), })), 1, ); @@ -116,6 +121,7 @@ export async function handleCallback( query , db , auth , + readModels = loadModelCatalog, ) { let answer = "Action handled"; try { @@ -131,7 +137,7 @@ export async function handleCallback( } if (action.type === "model_select") { - answer = await handleModelSelect(api, query, db, action.payload); + answer = await handleModelSelect(api, query, db, action.payload, readModels); return; } if (action.type === "effort_select") { @@ -222,15 +228,17 @@ async function handleModelSelect( query , db , payload , + readModels , ) { const parsed = parsePayload(payload); if (!parsed) return "Invalid model selection"; const binding = db.getBinding(parsed.bindingId); if (!binding) return "Binding not found"; - const catalog = loadModelCatalog(); - const entry = catalog[Number(parsed.value)]; - if (!entry) return "Model not found"; + const catalog = await readModels(); + const matches = catalog.filter(entry => modelToken(entry.id) === parsed.value); + if (matches.length !== 1) return "Model selection expired. Open the model picker again."; + const entry = matches[0]; db.setBindingModel(parsed.bindingId, entry.id); await sendCallbackMessage(api, query, `Model set to ${entry.id}`); diff --git a/plugins/codexclaw/components/messenger-bridge/dist/win-exec.js b/plugins/codexclaw/components/messenger-bridge/dist/win-exec.js index 2e778e07..a6ef0f02 100644 --- a/plugins/codexclaw/components/messenger-bridge/dist/win-exec.js +++ b/plugins/codexclaw/components/messenger-bridge/dist/win-exec.js @@ -1,89 +1,2 @@ -/** - * win-exec.ts - one entry point for spawning external commands. - * - * Three Windows facts drive this: - * 1. npm-installed CLIs are `.cmd` shims, and Node refuses shell-less `.cmd` spawns - * after the CVE-2024-27980 hardening. - * 2. A bare command name skips PATHEXT resolution, so `spawn("npm")` ENOENTs even - * with `npm.cmd` on PATH (measured, 002 B4). - * 3. `shell: true` is not a fix: Node does not escape cmd metacharacters there, so - * a path containing `&` or `^` becomes a command injection. - * - * Env vars are read case-insensitively: a spawned child can arrive with `Path`, - * `PATH`, or both, and reading one fixed spelling resolves against the wrong list - * (001 3.2). - */ -import { existsSync } from "node:fs"; -import { isAbsolute, join } from "node:path"; - - - - - - - -/** Case-insensitive env lookup with a key-scan fallback. */ -export function envValue(env , name ) { - const direct = env[name]; - if (direct !== undefined) return direct; - const lower = name.toLowerCase(); - for (const key of Object.keys(env)) { - if (key.toLowerCase() === lower) return env[key]; - } - return undefined; -} - -const DEFAULT_PATHEXT = ".COM;.EXE;.BAT;.CMD"; - -/** Resolve `command` against PATH + PATHEXT. Returns the input when nothing matches. */ -export function resolveWindowsCommand(command , env ) { - if (command.includes("/") || command.includes("\\") || isAbsolute(command)) return command; - const exts = (envValue(env, "PATHEXT") ?? DEFAULT_PATHEXT).split(";").filter((e) => e.length > 0); - // A WINDOWS PATH is always ";"-separated. node:path's `delimiter` follows the - // HOST, so on a Linux runner exercising this win32-only walk it would be ":" - // and the whole PATH would collapse into one bogus directory entry. - const dirs = (envValue(env, "PATH") ?? "").split(";").filter((d) => d.length > 0); - for (const dir of dirs) { - for (const ext of exts) { - const candidate = join(dir, command + ext); - if (existsSync(candidate)) return candidate; - // Case-sensitive filesystems (WSL, Linux CI): PATHEXT spells ".EXE" but - // real shims are "npm.cmd" / "gh.exe". Retry the lowercased extension. - const lowered = join(dir, command + ext.toLowerCase()); - if (existsSync(lowered)) return lowered; - } - } - return command; -} - -/** cross-spawn's escaping: double backslashes before quotes, quote, then caret. */ -function escapeCmdArg(arg ) { - let out = arg.replace(/(\\*)"/g, '$1$1\\"').replace(/(\\*)$/, "$1$1"); - out = `"${out}"`; - return out.replace(/[()%!^"<>&|;, ]/g, "^$&"); -} - -function escapeCmdCommand(command ) { - return command.replace(/[()%!^"<>&|;, ]/g, "^$&"); -} - -/** - * Build the spawn shape for `command`. POSIX is a passthrough; win32 resolves - * PATHEXT and routes only `.cmd`/`.bat` through cmd.exe. - */ -export function commandInvocation( - command , - args , - platform = process.platform, - env = process.env, -) { - if (platform !== "win32") return { file: command, args: [...args], options: {} }; - const resolved = resolveWindowsCommand(command, env); - if (!/\.(cmd|bat)$/i.test(resolved)) return { file: resolved, args: [...args], options: {} }; - const line = [escapeCmdCommand(resolved), ...args.map(escapeCmdArg)].join(" "); - return { - file: envValue(env, "ComSpec") ?? "cmd.exe", - args: ["/d", "/s", "/c", `"${line}"`], - options: { windowsVerbatimArguments: true }, - }; -} +/** Compatibility export; subprocess escaping is shared with live model discovery. */ +export * from "../../subagent-config/dist/win-exec.js"; diff --git a/plugins/codexclaw/components/messenger-bridge/src/api-compat.ts b/plugins/codexclaw/components/messenger-bridge/src/api-compat.ts index 45aae97c..8198d5c8 100644 --- a/plugins/codexclaw/components/messenger-bridge/src/api-compat.ts +++ b/plugins/codexclaw/components/messenger-bridge/src/api-compat.ts @@ -6,16 +6,16 @@ * the Vite dev middleware. When cxc serve hosts the built GUI statically those * routes must exist or role saves silently fail (A-audit finding 2). * - * Source of truth for the route semantics: gui/src/server/middleware.ts + - * gui/src/server/handlers.ts. This module mirrors them over the already + * Subagent settings semantics are shared with the Vite handlers through + * subagent-config/settings-api. Other routes mirror them over the already * COMPILED component dists (relative .js specifiers survive the build's * .ts→.js rewrite untouched and resolve identically from src/ and dist/). * Phase 6 unifies the GUI dev middleware onto this module. */ import { spawnSync } from "node:child_process"; // Compiled component dists — runtime-typed, so minimal local shapes below. -import { readConfig, setRole, ROLES } from "../../subagent-config/dist/store.js"; -import { buildCatalog } from "../../subagent-config/dist/catalog.js"; +import { getSettings, updateSettings, settingsResponse } from "../../subagent-config/dist/settings-api.js"; +import { readCatalog } from "../../subagent-config/dist/live-catalog.js"; import { detectOcx } from "../../provider-bridge/dist/detect.js"; import type { ApiRoute, ApiResponse } from "./server.ts"; import { splitLines } from "./text-lines.ts"; @@ -51,41 +51,8 @@ function detectDeps(): Record { }; } -/** Map provider detection to the catalog's provider input — mirrored from gui/src/server/handlers.ts. */ -function providerToCatalogInput(status: ProviderStatusShape): Record { - if (status.mode === "provider") { - // ocx-synced models surface via the native config cache, not a live call. - return { mode: "provider", ocxModels: undefined }; - } - return { mode: status.mode === "error" ? "error" : "native" }; -} - -function getSubagentsRoute(cwd: string): ApiResponse { - return { status: 200, body: readConfig(cwd) }; -} - -function postSubagentsRoute(cwd: string, body: unknown): ApiResponse { - if (!body || typeof body !== "object") return { status: 400, body: { error: "missing body" } }; - const b = body as Record; - const role = b.role as (typeof ROLES)[number]; - if (!ROLES.includes(role)) { - return { status: 400, body: { error: `unknown role "${String(b.role)}"` } }; - } - const patch: Record = {}; - if (b.mode !== undefined) patch.mode = b.mode; - if (b.model !== undefined) patch.model = b.model; - if (b.promptOverride !== undefined) patch.promptOverride = b.promptOverride; - try { - return { status: 200, body: setRole(cwd, role, patch) }; - } catch (err) { - return { status: 400, body: { error: err instanceof Error ? err.message : String(err) } }; - } -} - -function getCatalogRoute(): ApiResponse { - const status = detectOcx(detectDeps()) as ProviderStatusShape; - const catalog = buildCatalog({ providerStatus: providerToCatalogInput(status) }); - return { status: 200, body: catalog }; +async function getCatalogRoute(forceRefresh = false): Promise { + return { status: 200, body: await readCatalog({ forceRefresh }) }; } function getProviderRoute(): ApiResponse { @@ -100,14 +67,14 @@ export function apiCompatRoutes(): ApiRoute[] { { method: "GET", path: "/api/subagents", - handler: (ctx) => getSubagentsRoute(ctx.cwd), + handler: (ctx, _body, url) => settingsResponse(() => getSettings(ctx.cwd, url.searchParams.get("scope") ?? undefined)), }, { method: "POST", path: "/api/subagents", - handler: (ctx, body) => postSubagentsRoute(ctx.cwd, body), + handler: (ctx, body) => settingsResponse(() => updateSettings(ctx.cwd, body)), }, - { method: "GET", path: "/api/catalog", handler: () => getCatalogRoute() }, + { method: "GET", path: "/api/catalog", handler: (_ctx, _body, url) => getCatalogRoute(url.searchParams.get("refresh") === "1") }, { method: "GET", path: "/api/provider", handler: () => getProviderRoute() }, ]; } diff --git a/plugins/codexclaw/components/messenger-bridge/src/gateway-commands.ts b/plugins/codexclaw/components/messenger-bridge/src/gateway-commands.ts index 2c0efb76..a755e340 100644 --- a/plugins/codexclaw/components/messenger-bridge/src/gateway-commands.ts +++ b/plugins/codexclaw/components/messenger-bridge/src/gateway-commands.ts @@ -8,7 +8,7 @@ import { realpathSync, statSync } from "node:fs"; import { homedir } from "node:os"; import { resolve } from "node:path"; -import { buildCatalog } from "../../subagent-config/dist/catalog.js"; +import { readCatalog } from "../../subagent-config/dist/live-catalog.js"; import type { AgentService, IncomingRequest, IncomingResult } from "./agent-service.ts"; import type { ApprovalDecision } from "./approval-relay.ts"; import { chunkEmbedDescription, type DiscordEmbed } from "./discord-api.ts"; @@ -29,6 +29,7 @@ export interface GatewayCommandContext { onApprovalRequest?: IncomingRequest["onApprovalRequest"]; onEvent?: IncomingRequest["onEvent"]; now?: () => Date; + readModelCatalog?: () => Promise<{ entries: Array<{ id: string; label: string; source: string }> }>; } export interface GatewayCommandResult { @@ -315,7 +316,7 @@ async function handleModel(ctx: GatewayCommandContext): Promise }; +async function modelListResult(reader: NonNullable = readCatalog): Promise { + const catalog = await reader() as { entries?: Array<{ id?: unknown; label?: unknown; source?: unknown }> }; const groups = groupCatalogEntries(catalog.entries ?? []); const text = groups.length === 0 ? "No models found." diff --git a/plugins/codexclaw/components/messenger-bridge/src/telegram-commands.ts b/plugins/codexclaw/components/messenger-bridge/src/telegram-commands.ts index 4e1aac4b..9cf20fce 100644 --- a/plugins/codexclaw/components/messenger-bridge/src/telegram-commands.ts +++ b/plugins/codexclaw/components/messenger-bridge/src/telegram-commands.ts @@ -162,7 +162,7 @@ async function handleModel(ctx: CommandContext): Promise { const current = String(result?.data?.model ?? "default"); return { text: result?.text ?? `Current model: ${current}`, - keyboard: buildModelPicker(loadModelCatalog(), current, binding.id), + keyboard: buildModelPicker(await loadModelCatalog(), current, binding.id), }; } diff --git a/plugins/codexclaw/components/messenger-bridge/src/telegram-interactive.ts b/plugins/codexclaw/components/messenger-bridge/src/telegram-interactive.ts index fdb95598..92d604ba 100644 --- a/plugins/codexclaw/components/messenger-bridge/src/telegram-interactive.ts +++ b/plugins/codexclaw/components/messenger-bridge/src/telegram-interactive.ts @@ -2,9 +2,10 @@ * telegram-interactive.ts — inline keyboard callback encoding and dispatch. * * Telegram callback_data is capped at 64 bytes, so payloads stay compact and - * model selections use catalog indexes instead of full model ids. + * model selections use stable ID hashes instead of mutable catalog indexes. */ -import { buildCatalog } from "../../subagent-config/dist/catalog.js"; +import { readCatalog } from "../../subagent-config/dist/live-catalog.js"; +import { createHash } from "node:crypto"; import { AGENT_EFFORTS, AGENT_THREAD_MODES, AGENT_TOOL_PROGRESS_MODES, type BridgeDb } from "./db.ts"; import { telegramReplyThreadId, telegramTopicId, type TelegramApi, type TgCallbackQuery } from "./telegram-api.ts"; import type { InlineKeyboard } from "./telegram-commands.ts"; @@ -61,8 +62,8 @@ export function decodeCallback(data: string): CallbackAction | null { return { type, payload: rest.join(":") }; } -export function loadModelCatalog(): CatalogEntry[] { - const catalog = buildCatalog() as { entries?: Array<{ id?: unknown; label?: unknown }> }; +export async function loadModelCatalog(): Promise { + const catalog = await readCatalog() as { entries?: Array<{ id?: unknown; label?: unknown }> }; const entries = Array.isArray(catalog.entries) ? catalog.entries : []; return entries .filter((entry): entry is { id: string; label?: string } => typeof entry.id === "string" && entry.id.length > 0) @@ -72,11 +73,15 @@ export function loadModelCatalog(): CatalogEntry[] { })); } +export function modelToken(id: string): string { + return "h" + createHash("sha256").update(id).digest("hex").slice(0, 24); +} + export function buildModelPicker(catalog: CatalogEntry[], current: string, bindingId = 0): InlineKeyboard { return rows( - catalog.map((entry, index) => ({ + catalog.map((entry) => ({ text: `${entry.id === current ? "* " : ""}${entry.label ?? entry.id}`, - callback_data: encodeCallback({ type: "model_select", payload: `${bindingId}:${index}` }), + callback_data: encodeCallback({ type: "model_select", payload: `${bindingId}:${modelToken(entry.id)}` }), })), 1, ); @@ -116,6 +121,7 @@ export async function handleCallback( query: TgCallbackQuery, db: BridgeDb, auth: CallbackAuthContext, + readModels: () => Promise = loadModelCatalog, ): Promise { let answer = "Action handled"; try { @@ -131,7 +137,7 @@ export async function handleCallback( } if (action.type === "model_select") { - answer = await handleModelSelect(api, query, db, action.payload); + answer = await handleModelSelect(api, query, db, action.payload, readModels); return; } if (action.type === "effort_select") { @@ -222,15 +228,17 @@ async function handleModelSelect( query: TgCallbackQuery, db: BridgeDb, payload: string, + readModels: () => Promise, ): Promise { const parsed = parsePayload(payload); if (!parsed) return "Invalid model selection"; const binding = db.getBinding(parsed.bindingId); if (!binding) return "Binding not found"; - const catalog = loadModelCatalog(); - const entry = catalog[Number(parsed.value)]; - if (!entry) return "Model not found"; + const catalog = await readModels(); + const matches = catalog.filter(entry => modelToken(entry.id) === parsed.value); + if (matches.length !== 1) return "Model selection expired. Open the model picker again."; + const entry = matches[0]; db.setBindingModel(parsed.bindingId, entry.id); await sendCallbackMessage(api, query, `Model set to ${entry.id}`); diff --git a/plugins/codexclaw/components/messenger-bridge/src/win-exec.ts b/plugins/codexclaw/components/messenger-bridge/src/win-exec.ts index 5758cd06..a6ef0f02 100644 --- a/plugins/codexclaw/components/messenger-bridge/src/win-exec.ts +++ b/plugins/codexclaw/components/messenger-bridge/src/win-exec.ts @@ -1,89 +1,2 @@ -/** - * win-exec.ts - one entry point for spawning external commands. - * - * Three Windows facts drive this: - * 1. npm-installed CLIs are `.cmd` shims, and Node refuses shell-less `.cmd` spawns - * after the CVE-2024-27980 hardening. - * 2. A bare command name skips PATHEXT resolution, so `spawn("npm")` ENOENTs even - * with `npm.cmd` on PATH (measured, 002 B4). - * 3. `shell: true` is not a fix: Node does not escape cmd metacharacters there, so - * a path containing `&` or `^` becomes a command injection. - * - * Env vars are read case-insensitively: a spawned child can arrive with `Path`, - * `PATH`, or both, and reading one fixed spelling resolves against the wrong list - * (001 3.2). - */ -import { existsSync } from "node:fs"; -import { isAbsolute, join } from "node:path"; - -export interface Invocation { - file: string; - args: string[]; - options: { windowsVerbatimArguments?: boolean }; -} - -/** Case-insensitive env lookup with a key-scan fallback. */ -export function envValue(env: NodeJS.ProcessEnv, name: string): string | undefined { - const direct = env[name]; - if (direct !== undefined) return direct; - const lower = name.toLowerCase(); - for (const key of Object.keys(env)) { - if (key.toLowerCase() === lower) return env[key]; - } - return undefined; -} - -const DEFAULT_PATHEXT = ".COM;.EXE;.BAT;.CMD"; - -/** Resolve `command` against PATH + PATHEXT. Returns the input when nothing matches. */ -export function resolveWindowsCommand(command: string, env: NodeJS.ProcessEnv): string { - if (command.includes("/") || command.includes("\\") || isAbsolute(command)) return command; - const exts = (envValue(env, "PATHEXT") ?? DEFAULT_PATHEXT).split(";").filter((e) => e.length > 0); - // A WINDOWS PATH is always ";"-separated. node:path's `delimiter` follows the - // HOST, so on a Linux runner exercising this win32-only walk it would be ":" - // and the whole PATH would collapse into one bogus directory entry. - const dirs = (envValue(env, "PATH") ?? "").split(";").filter((d) => d.length > 0); - for (const dir of dirs) { - for (const ext of exts) { - const candidate = join(dir, command + ext); - if (existsSync(candidate)) return candidate; - // Case-sensitive filesystems (WSL, Linux CI): PATHEXT spells ".EXE" but - // real shims are "npm.cmd" / "gh.exe". Retry the lowercased extension. - const lowered = join(dir, command + ext.toLowerCase()); - if (existsSync(lowered)) return lowered; - } - } - return command; -} - -/** cross-spawn's escaping: double backslashes before quotes, quote, then caret. */ -function escapeCmdArg(arg: string): string { - let out = arg.replace(/(\\*)"/g, '$1$1\\"').replace(/(\\*)$/, "$1$1"); - out = `"${out}"`; - return out.replace(/[()%!^"<>&|;, ]/g, "^$&"); -} - -function escapeCmdCommand(command: string): string { - return command.replace(/[()%!^"<>&|;, ]/g, "^$&"); -} - -/** - * Build the spawn shape for `command`. POSIX is a passthrough; win32 resolves - * PATHEXT and routes only `.cmd`/`.bat` through cmd.exe. - */ -export function commandInvocation( - command: string, - args: string[], - platform: NodeJS.Platform = process.platform, - env: NodeJS.ProcessEnv = process.env, -): Invocation { - if (platform !== "win32") return { file: command, args: [...args], options: {} }; - const resolved = resolveWindowsCommand(command, env); - if (!/\.(cmd|bat)$/i.test(resolved)) return { file: resolved, args: [...args], options: {} }; - const line = [escapeCmdCommand(resolved), ...args.map(escapeCmdArg)].join(" "); - return { - file: envValue(env, "ComSpec") ?? "cmd.exe", - args: ["/d", "/s", "/c", `"${line}"`], - options: { windowsVerbatimArguments: true }, - }; -} +/** Compatibility export; subprocess escaping is shared with live model discovery. */ +export * from "../../subagent-config/dist/win-exec.js"; diff --git a/plugins/codexclaw/components/messenger-bridge/test/gateway-commands.test.ts b/plugins/codexclaw/components/messenger-bridge/test/gateway-commands.test.ts index 32e19e37..93724321 100644 --- a/plugins/codexclaw/components/messenger-bridge/test/gateway-commands.test.ts +++ b/plugins/codexclaw/components/messenger-bridge/test/gateway-commands.test.ts @@ -74,7 +74,7 @@ test("/model reserved subargs list/reset are handled before verbatim model stora db.setBindingModel(binding.id, "gpt-custom"); const base = { bindingId: binding.id, db, agentService: stubAgent(), agentId: agent.id, args: "" }; - const list = await dispatchGatewayCommand("model", { ...base, args: "list" }); + const list = await dispatchGatewayCommand("model", { ...base, args: "list", readModelCatalog: async () => ({ entries: ["gpt-5.5", "anthropic/claude-sonnet-5"].map(id => ({ id, label: id, source: id.includes("/") ? "ocx" : "native" })) }) }); assert.match(list?.text ?? "", /Available models/); assert.match(list?.telegramHtml ?? "", /anthropic\/claude-sonnet-5/); assert.ok((list?.telegramHtmlChunks as string[]).every((chunk) => chunk.length <= 4096)); @@ -108,6 +108,7 @@ test("/model list emits provider continuation fields before Discord cap overflow agentService: stubAgent(), agentId: null, args: "list", + readModelCatalog: async () => ({ entries: [...bigProviderIds, ...overflowProviderIds].map(id => ({ id, label: id, source: "ocx" })) }), }); const embed = result?.discordEmbed; const serialized = JSON.stringify(embed); diff --git a/plugins/codexclaw/components/messenger-bridge/test/subagent-effort.test.ts b/plugins/codexclaw/components/messenger-bridge/test/subagent-effort.test.ts new file mode 100644 index 00000000..485581f2 --- /dev/null +++ b/plugins/codexclaw/components/messenger-bridge/test/subagent-effort.test.ts @@ -0,0 +1,124 @@ +/** cxc serve API persistence contract, including a fresh server process. */ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdtempSync, readFileSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { spawn } from "node:child_process"; +import { once } from "node:events"; + +async function startServer(cwd: string) { + const serverModule = new URL("../src/server.ts", import.meta.url).href; + const dbModule = new URL("../src/db.ts", import.meta.url).href; + const script = ` + import { createBridgeServer } from ${JSON.stringify(serverModule)}; + import { openBridgeDb } from ${JSON.stringify(dbModule)}; + const cwd = process.argv[1]; + const db = openBridgeDb(cwd); + const server = createBridgeServer({ cwd, db, version: 'effort-test' }); + server.listen(0, '127.0.0.1', () => console.log(server.address().port)); + process.on('SIGTERM', () => server.close(() => { db.close(); process.exit(0); })); + `; + const child = spawn(process.execPath, ["--input-type=module", "-e", script, cwd], { stdio: ["ignore", "pipe", "pipe"], env: { ...process.env, CODEX_HOME: join(cwd, "test-codex-home"), CODEXCLAW_HOME: join(cwd, "test-cxc-home") } }); + let stderr = ""; + child.stderr.on("data", chunk => { stderr += chunk; }); + const exit = once(child, "exit"); + const stop = async () => { child.kill(); await exit; }; + try { + const ready = once(child.stdout, "data", { signal: AbortSignal.timeout(10_000) }); + const [data] = await Promise.race([ready, exit.then(() => { throw new Error(`server exited before listening: ${stderr}`); })]); + const port = Number(String(data).trim()); + assert.ok(port > 0, `invalid server port: ${String(data)} ${stderr}`); + return { base: `http://127.0.0.1:${port}`, stop }; + } catch (err) { await stop(); throw err; } +} + +function post(base: string, body: unknown) { + return fetch(`${base}/api/subagents`, { + method: "POST", headers: { "content-type": "application/json", "x-codexclaw-local": "1" }, body: JSON.stringify(body), + }); +} + +test("serve persists effort across GET and process restart, preserves models, accepts null and rejects invalid values", async () => { + const cwd = mkdtempSync(join(tmpdir(), "cxc-effort-")); + let server = await startServer(cwd); + try { + const models = { explorer: "fixture-luna", reviewer: "fixture-sol", executor: "fixture-terra" }; + for (const [role, model] of Object.entries(models)) { + assert.equal((await post(server.base, { role, mode: "model", model, promptOverride: `${role} prompt` })).status, 200); + } + const initial = await (await fetch(`${server.base}/api/subagents`)).json(); + const snapshot = (effort: string | null) => ({ + ...initial, roles: { ...initial.roles, explorer: { ...initial.roles.explorer, effort } }, + }); + // These are isolated fixture values, never user preferences. + for (const effort of ["low", "medium", "high", "xhigh"]) { + const res = await post(server.base, { role: "explorer", effort }); + assert.equal(res.status, 200); + assert.deepEqual(await res.json(), snapshot(effort)); + assert.deepEqual(await (await fetch(`${server.base}/api/subagents`)).json(), snapshot(effort)); + } + const storePath = join(cwd, ".codexclaw/subagents.json"); + assert.deepEqual(JSON.parse(readFileSync(storePath, "utf8")), { roles: snapshot("xhigh").roles }); + await server.stop(); + server = await startServer(cwd); + assert.deepEqual(await (await fetch(`${server.base}/api/subagents`)).json(), snapshot("xhigh")); + // An omitted effort must preserve the saved value. + assert.deepEqual(await (await post(server.base, { role: "explorer", model: models.explorer })).json(), snapshot("xhigh")); + for (const effort of ["invalid", "", "HIGH", 3, false, {}, []]) { + const before = readFileSync(storePath, "utf8"); + const res = await post(server.base, { role: "explorer", effort, promptOverride: "must not be saved" }); + assert.equal(res.status, 400, JSON.stringify(effort)); + assert.match((await res.json()).error, /invalid effort/); + assert.equal(readFileSync(storePath, "utf8"), before); + } + const reset = await post(server.base, { role: "explorer", effort: null }); + assert.equal(reset.status, 200); + assert.deepEqual(await reset.json(), snapshot(null)); + await server.stop(); + server = await startServer(cwd); + assert.deepEqual(await (await fetch(`${server.base}/api/subagents`)).json(), snapshot(null)); + } finally { await server.stop(); rmSync(cwd, { recursive: true, force: true }); } +}); + +test('serve global defaults and project overrides survive restart; reset restores the next scope', async () => { + const cwd = mkdtempSync(join(tmpdir(), 'cxc-global-api-')); + let server = await startServer(cwd); + const get = async (scope = 'project') => (await fetch(`${server.base}/api/subagents?scope=${scope}`)).json(); + try { + assert.equal((await get()).sources.explorer, 'session'); + for (const effort of ['low', 'medium', 'high', 'xhigh', null]) { + const res = await post(server.base, { scope: 'global', role: 'explorer', mode: 'model', model: 'global-luna', effort }); + assert.equal(res.status, 200); + assert.equal((await res.json()).roles.explorer.effort, effort); + await server.stop(); server = await startServer(cwd); + assert.equal((await get()).roles.explorer.effort, effort); + assert.equal((await get()).sources.explorer, 'global'); + } + await post(server.base, { role: 'explorer', model: 'project-luna', effort: null }); + await post(server.base, { scope: 'global', role: 'explorer', effort: 'high' }); + await server.stop(); server = await startServer(cwd); + const local = await get(); + assert.equal(local.roles.explorer.model, 'project-luna'); + assert.equal(local.roles.explorer.effort, null); + assert.equal(local.sources.explorer, 'project'); + assert.equal((await get('global')).roles.explorer.effort, 'high'); + const globalPath = join(cwd, 'test-cxc-home/subagents.json'); + const before = readFileSync(globalPath, 'utf8'); + for (const body of [ + { scope: 'global', effort: 'invalid' }, { scope: 'global', effort: {} }, + { scope: '../escape', effort: 'low' }, { scope: null, effort: 'low' }, + { scope: 'global', inherit: 'yes' }, { scope: 'global', inherit: true, effort: 'low' }, + ]) { + assert.equal((await post(server.base, { role: 'explorer', ...body })).status, 400); + assert.equal(readFileSync(globalPath, 'utf8'), before); + } + assert.equal((await fetch(`${server.base}/api/subagents?scope=bad`)).status, 400); + assert.equal((await post(server.base, { role: 'explorer', inherit: true })).status, 200); + assert.equal((await get()).roles.explorer.effort, 'high'); + assert.equal((await post(server.base, { scope: 'global', role: 'explorer', inherit: true })).status, 200); + await server.stop(); server = await startServer(cwd); + assert.equal((await get()).sources.explorer, 'session'); + assert.equal((await get()).roles.explorer.effort, null); + } finally { await server.stop(); rmSync(cwd, { recursive: true, force: true }); } +}); diff --git a/plugins/codexclaw/components/messenger-bridge/test/telegram-interactive.test.ts b/plugins/codexclaw/components/messenger-bridge/test/telegram-interactive.test.ts index b9c4eb6d..971d0ce7 100644 --- a/plugins/codexclaw/components/messenger-bridge/test/telegram-interactive.test.ts +++ b/plugins/codexclaw/components/messenger-bridge/test/telegram-interactive.test.ts @@ -8,6 +8,7 @@ import { openBridgeDb } from "../src/db.ts"; import { buildEffortPicker, buildModelPicker, + modelToken, buildToolProgressPicker, decodeCallback, encodeCallback, @@ -131,14 +132,20 @@ test("handleCallback updates model and always answers the callback", async () => try { const agent = db.createAgent("telegram-1", "telegram", "tok"); const binding = db.getOrCreateAgentBinding(agent.id, "telegram", "500", cwd); - const firstModel = loadModelCatalog()[0]?.id; + const firstModel = "fixture/model-a"; assert.ok(firstModel, "model catalog should not be empty"); const api = mockApi(); - await handleCallback(api, callback(encodeCallback({ type: "model_select", payload: `${binding.id}:0` })), db, allowAgent(agent.id)); + await handleCallback(api, callback(encodeCallback({ type: "model_select", payload: `${binding.id}:${modelToken(firstModel)}` })), db, allowAgent(agent.id), async () => [{id: "fixture/model-b"}, {id: firstModel}]); assert.equal(db.getAgent(agent.id)?.model, "default"); assert.equal(db.getBinding(binding.id)?.model, firstModel); + // Old positional callbacks expire rather than selecting a different live row. + await handleCallback(api, callback(encodeCallback({type:"model_select",payload:`${binding.id}:0`})), db, allowAgent(agent.id), async()=>[{id:"wrong-model"}]); + assert.equal(db.getBinding(binding.id)?.model, firstModel); + const rendered = buildModelPicker([{id:firstModel},{id:"fixture/model-b"}],firstModel,binding.id); + await handleCallback(api,callback(rendered[1][0].callback_data!),db,allowAgent(agent.id),async()=>[{id:"fixture/model-b"},{id:firstModel}]); + assert.equal(db.getBinding(binding.id)?.model,"fixture/model-b"); assert.ok(api.calls.some((call) => call.method === "sendMessage")); assert.ok(api.calls.some((call) => call.method === "answerCallbackQuery")); } finally { diff --git a/plugins/codexclaw/components/subagent-config/dist/catalog.js b/plugins/codexclaw/components/subagent-config/dist/catalog.js index d24ff14e..bb81a9bf 100644 --- a/plugins/codexclaw/components/subagent-config/dist/catalog.js +++ b/plugins/codexclaw/components/subagent-config/dist/catalog.js @@ -1,24 +1,7 @@ -/** - * catalog.ts — selectable model catalog (L25 / 250-252). - * - * Source = Codex-native catalog (always) + ocx-backed models (when ocx is - * detected and exposes a catalog). Native entries come first; entries are - * deduplicated by stable model id keeping the native one. No network fetch, no - * vendored ocx files, no selected-model persistence (L24 owns that). - * - * Native source: the Codex live catalog cache at CODEX_MODELS_CACHE_PATH, read - * through an allowlist. When the cache is absent/unreadable, fall back to the - * documented NATIVE_OPENAI_MODELS set (opencodex src/codex-catalog.ts:44). - * - * Slug parity (L9.2 / 092): the LIVE Codex catalog keys each entry by `slug` - * (bare like "gpt-5.5", or routed "provider/model"), not `id` (opencodex - * codex-catalog.ts:152,183). The cache reader therefore accepts BOTH `id` and - * `slug`, and dedup compares on the resolved key so a native slug and an ocx id - * for the same model collapse, native kept first. - */ +/** Native catalog parsing and pure catalog composition. Live OCX discovery lives in live-catalog.ts. */ import { existsSync, readFileSync } from "node:fs"; import { homedir } from "node:os"; -import { join } from "node:path"; +import { isAbsolute, join, resolve } from "node:path"; export const NATIVE_OPENAI_MODELS = ["gpt-5.5", "gpt-5.4", "gpt-5.4-mini", "gpt-5.6-luna"] ; @@ -37,6 +20,7 @@ export const NATIVE_OPENAI_MODELS = ["gpt-5.5", "gpt-5.4", "gpt-5.4-mini", "gpt- + /** Provider status as exposed by the L23 bridge (subset this loop needs). */ @@ -71,47 +55,62 @@ function isRoutedSlug(key ) { return key.includes("/"); } -/** Read the Codex live catalog cache (CODEX_MODELS_CACHE_PATH) through the - * allowlist. Reads each entry by `id` OR `slug` (live catalog uses slug). - * Returns ids or null when absent/unreadable. - * - * L20/WP4: the cache is the codex config catalog, which opencodex SYNCS its - * routed `provider/model` slugs into. codexclaw reads that config (it never - * calls ocx directly). So the allowlist admits BOTH the documented native ids - * AND any routed slug (contains "/") — dropping routed slugs would hide exactly - * the ocx-synced models the subagent config is meant to select. */ -export function readNativeCacheDefault(env = process.env) { - // Resolve like opencodex (codex-paths.ts:30): explicit override, else - // $CODEX_HOME/models_cache.json, else ~/.codex/models_cache.json. Nothing in - // `cxc serve` sets CODEX_MODELS_CACHE_PATH, so the homedir default is what - // makes the ocx-synced routed slugs actually load in practice. - const path = - env.CODEX_MODELS_CACHE_PATH ?? - join(env.CODEX_HOME ?? join(homedir(), ".codex"), "models_cache.json"); - if (!existsSync(path)) return null; +/** Root-level TOML path only: never read a similarly named key inside a table. */ +export function nativeCatalogPath(env = process.env) { + if (env.CODEX_MODELS_CACHE_PATH?.trim()) return env.CODEX_MODELS_CACHE_PATH; + const home = env.CODEX_HOME?.trim() || join(homedir(), ".codex"); + try { + const lines = readFileSync(join(home, "config.toml"), "utf8").split(/\r?\n/); + for (const line of lines) { + if (/^\s*\[/.test(line)) break; + if (!/^\s*(?:model_catalog_json|"model_catalog_json"|'model_catalog_json')\s*=/.test(line)) continue; + const match = /^\s*(?:model_catalog_json|"model_catalog_json"|'model_catalog_json')\s*=\s*("(?:\\.|[^"\\])*"|'[^']*')\s*(?:#.*)?$/.exec(line); + if (!match) return null; + const value = match[1].startsWith("'") ? match[1].slice(1, -1) : JSON.parse(match[1]); + if (!value.trim()) return null; + const expanded = value.startsWith("~/") || value.startsWith("~\\") ? join(homedir(), value.slice(2)) : value; + return isAbsolute(expanded) ? expanded : resolve(home, expanded); + } + } catch (error) { + if ((error ).code !== "ENOENT") return null; + } + return join(home, "models_cache.json"); +} + +export function reasoningEfforts(raw ) { + if (!Array.isArray(raw)) return null; + return [...new Set(raw.flatMap(value => { + const effort = typeof value === "string" ? value : value && typeof value === "object" ? (value ).effort : undefined; + return typeof effort === "string" && effort.length > 0 ? [effort] : []; + }))]; +} + +export function readNativeCatalog(env = process.env) { + const path = nativeCatalogPath(env); + if (!path || !existsSync(path)) return null; try { - const parsed = JSON.parse(readFileSync(path, "utf8")) ; - const list = Array.isArray(parsed) ? parsed : (parsed )?.models; + const parsed = JSON.parse(readFileSync(path, "utf8")); + const list = Array.isArray(parsed) ? parsed : (parsed )?.models; if (!Array.isArray(list)) return null; - const ids = list.map(entryKey).filter((x) => typeof x === "string"); - // allowlist: ship documented native ids AND routed provider/model slugs - // (the ocx-synced entries). Dedup preserves first-seen order so a slug+id - // duplicate yields one entry. const seen = new Set (); - const allowed = ids.filter( - (id) => - ((NATIVE_OPENAI_MODELS ).includes(id) || isRoutedSlug(id)) && - !seen.has(id) && - (seen.add(id), true), - ); - return allowed.length ? allowed : null; - } catch { - return null; - } + return list.flatMap(raw => { + const id = entryKey(raw); + const row = raw && typeof raw === "object" ? raw : {}; + if (!id?.trim() || seen.has(id) || row.disabled === true || row.visibility === "hide") return []; + seen.add(id); + const source = isRoutedSlug(id) ? "ocx" : "native"; + return [{ id, source, label: id, reasoningEfforts: reasoningEfforts(row.reasoningEfforts ?? row.supported_reasoning_levels) }]; + }); + } catch { return null; } +} + +/** Legacy ID-only reader retained for pure consumers and fixtures. */ +export function readNativeCacheDefault(env = process.env) { + return readNativeCatalog(env)?.map(entry => entry.id) ?? null; } function nativeEntries(deps ) { - const ids = (deps.readNativeCache ?? readNativeCacheDefault)() ?? [...NATIVE_OPENAI_MODELS]; + const ids = (deps.readNativeCache ?? readNativeCacheDefault)() ?? []; // Entries from the codex config cache: bare ids are native; routed `provider/model` // slugs were synced in by opencodex, so label them as ocx-origin even though they // arrive through the native cache (codexclaw never calls ocx directly). @@ -131,7 +130,7 @@ export function buildCatalog(deps = {}) { const status = deps.providerStatus; if (!status || status.mode !== "provider") { - return { state: "native-catalog", entries: native }; + return { state: native.length ? "native-catalog" : "unavailable", entries: native }; } // ocx is active. If it exposes no catalog interface, the cache-sync channel may diff --git a/plugins/codexclaw/components/subagent-config/dist/cli.js b/plugins/codexclaw/components/subagent-config/dist/cli.js index b65de179..d5b696f8 100644 --- a/plugins/codexclaw/components/subagent-config/dist/cli.js +++ b/plugins/codexclaw/components/subagent-config/dist/cli.js @@ -13,7 +13,7 @@ * subagents set --mode default|model [--model ] [--effort |--clear-effort] * [--prompt |--clear-prompt] */ -import { readConfig, setRole, projectConfigTrustToken, ROLES, EFFORTS, } from "./store.js"; +import { readConfig, setRole, resetRole, projectConfigTrustToken, ROLES, EFFORTS, } from "./store.js"; import { realpathSync } from "node:fs"; import { fileURLToPath } from "node:url"; @@ -24,17 +24,30 @@ import { fileURLToPath } from "node:url"; + function isRole(v ) { return typeof v === "string" && (ROLES ).includes(v); } /** Pure structural parse of the `subagents` argv (excluding the leading verb). */ export function parseSubagentsArgs(argv ) { + // Scope is an explicit trailing selector, so prompt/model values stay literal. + if (argv.at(-1) === "--global" && !["--prompt", "--model"].includes(argv.at(-2) ?? "")) { + return { ...parseProjectArgs(argv.slice(0, -1)), scope: "global" }; + } + return parseProjectArgs(argv); +} + +function parseProjectArgs(argv ) { const sub = argv[0]; if (sub === undefined || sub === "list") return { action: "list" }; if (sub === "help" || sub === "--help" || sub === "-h") return { action: "help" }; if (sub === "trust-token") return { action: "trust-token" }; + if (sub === "reset") { + if (!isRole(argv[1]) || argv.length !== 2) return { action: "reset", error: "reset requires exactly one valid role" }; + return { action: "reset", role: argv[1] }; + } if (sub === "get") { if (!isRole(argv[1])) return { action: "get", error: `unknown role '${argv[1] ?? ""}' (expected ${ROLES.join("|")})` }; return { action: "get", role: argv[1] }; @@ -83,6 +96,8 @@ const HELP = [ " subagents list all role configs", " subagents get show one role config", " subagents set --mode default|model [--model ] [--effort |--clear-effort] [--prompt |--clear-prompt]", + " subagents reset remove the role override and inherit the next scope", + " Append --global to list/get/set/reset to manage user defaults", " subagents trust-token print an export bound to this repo and exact config", "", ` roles: ${ROLES.join(", ")}`, @@ -101,9 +116,9 @@ export function runSubagents(parsed , cwd ) case "help": return { code: 0, output: HELP }; case "list": - return { code: 0, output: JSON.stringify(readConfig(cwd), null, 2) }; + return { code: 0, output: JSON.stringify(readConfig(cwd, parsed.scope), null, 2) }; case "get": { - const cfg = readConfig(cwd); + const cfg = readConfig(cwd, parsed.scope); return { code: 0, output: JSON.stringify(cfg.roles[parsed.role ], null, 2) }; } case "trust-token": { @@ -111,9 +126,17 @@ export function runSubagents(parsed , cwd ) if (!token) return { code: 1, output: "subagents: cannot hash .codexclaw/subagents.json" }; return { code: 0, output: `export CODEXCLAW_TRUST_PROJECT_SUBAGENTS='${token}'` }; } + case "reset": { + try { + const cfg = resetRole(cwd, parsed.role , parsed.scope); + return { code: 0, output: JSON.stringify(cfg.roles[parsed.role ], null, 2) }; + } catch (err) { + return { code: 1, output: `subagents: ${err instanceof Error ? err.message : String(err)}` }; + } + } case "set": { try { - const cfg = setRole(cwd, parsed.role , parsed.patch ?? {}); + const cfg = setRole(cwd, parsed.role , parsed.patch ?? {}, parsed.scope); return { code: 0, output: JSON.stringify(cfg.roles[parsed.role ], null, 2) }; } catch (err) { return { code: 1, output: `subagents: ${err instanceof Error ? err.message : String(err)}` }; diff --git a/plugins/codexclaw/components/subagent-config/dist/live-catalog.js b/plugins/codexclaw/components/subagent-config/dist/live-catalog.js new file mode 100644 index 00000000..a69a08d5 --- /dev/null +++ b/plugins/codexclaw/components/subagent-config/dist/live-catalog.js @@ -0,0 +1,119 @@ +/** Read-only OCX discovery with a shared, bounded-age last-success cache. */ +import { execFile } from "node:child_process"; +import { createHash, randomUUID } from "node:crypto"; +import { mkdirSync, readFileSync, writeFileSync, rmSync } from "node:fs"; +import { join } from "node:path"; +import { readNativeCatalog, reasoningEfforts, } from "./catalog.js"; +import { cxcHome } from "./store.js"; +import { commandInvocation } from "./win-exec.js"; +import { renameWithRetry } from "./atomic-write.js"; + +export const CATALOG_TTL_MS = 30_000; + + + + + + + + + + + + + +const pending = new Map (); + +export function runOcxModels(env ) { + const invocation = commandInvocation("ocx", ["models", "live", "--json"], process.platform, env); + return new Promise((resolve, reject) => { + execFile(invocation.file, invocation.args, { + ...invocation.options, env, encoding: "utf8", timeout: 12_000, + killSignal: "SIGKILL", maxBuffer: 4 * 1024 * 1024, windowsHide: true, + }, (error, stdout) => error ? reject(error) : resolve(stdout)); + }); +} + +export function parseOcxModels(stdout ) { + const parsed = JSON.parse(stdout); + if (!Array.isArray(parsed)) throw new Error("invalid OCX catalog"); + const seen = new Set (); + return parsed.flatMap(raw => { + if (!raw || typeof raw !== "object" || Array.isArray(raw)) throw new Error("invalid OCX model row"); + const row = raw ; + if (row.disabled === true || row.initialSelectionPending === true) return []; + const id = typeof row.namespaced === "string" ? row.namespaced : row.native === true ? row.id : undefined; + if (typeof id !== "string" || !id.trim()) throw new Error("invalid OCX model id"); + if (seen.has(id)) return []; + seen.add(id); + return [{ id, source: row.native === true ? "native" : "ocx" , + label: typeof row.displayName === "string" && row.displayName.trim() ? `${row.displayName} (${id})` : id, + reasoningEfforts: reasoningEfforts(row.reasoningEfforts) }]; + }); +} + +function sourceKey(env ) { + return createHash("sha256").update(JSON.stringify([ + env.CODEX_HOME ?? "", env.CODEX_MODELS_CACHE_PATH ?? "", env.PATH ?? env.Path ?? "", env.OPENCODEX_HOME ?? "", + ])).digest("hex"); +} +function cachedCatalog(path , key , now ) { + try { + const raw = JSON.parse(readFileSync(path, "utf8")); + const c = raw.catalog ; + if (raw.key !== key || !c || c.status !== "fresh" || !["ocx", "native"].includes(c.source) || !Array.isArray(c.entries)) return null; + if (!c.fetchedAt || !Number.isFinite(Date.parse(c.fetchedAt)) || Date.parse(c.fetchedAt) > now + 1000) return null; + if (!c.entries.every(entry => entry && typeof entry.id === "string" && entry.id.length && typeof entry.label === "string" + && ["ocx", "native"].includes(entry.source) && (entry.reasoningEfforts === null || Array.isArray(entry.reasoningEfforts) && entry.reasoningEfforts.every(e => typeof e === "string")))) return null; + return c; + } catch { return null; } +} +function persist(path , key , catalog , home ) { + const temp = `${path}.${randomUUID()}.tmp`; + try { + mkdirSync(home, { recursive: true, mode: 0o700 }); + writeFileSync(temp, JSON.stringify({ key, catalog }) + "\n", { flag: "wx", mode: 0o600 }); + renameWithRetry(temp, path); + return true; + } catch { return false; } + finally { try { rmSync(temp, { force: true }); } catch { /* best-effort own temp cleanup */ } } +} + +/** Project cwd never participates in cache identity. Explicit CXC home isolates all state. */ +export async function readCatalog(options = {}) { + const env = options.env ?? process.env; + const now = options.now ?? Date.now; + const home = cxcHome(env); + const path = join(home, "model-catalog.json"); + const key = sourceKey(env); + const pendingKey = `${path}:${key}`; + const running = pending.get(pendingKey); + if (running) return running; + const cached = cachedCatalog(path, key, now()); + if (!options.forceRefresh && cached && now() - Date.parse(cached.fetchedAt ) < CATALOG_TTL_MS) return cached; + const query = async () => { + let source = "ocx"; + try { + let entries ; + try { entries = parseOcxModels(await (options.runOcx ?? runOcxModels)(env)); } + catch (error) { + if ((error ).code !== "ENOENT") throw error; + source = "native"; + const native = (options.readNative ?? readNativeCatalog)(env); + if (native === null) throw new Error("native catalog unavailable"); + entries = native; + } + const catalog = { state: source === "ocx" ? "ocx-active" : "native-catalog", entries, + status: "fresh", source, fetchedAt: new Date(now()).toISOString() }; + if (!persist(path, key, catalog, home)) catalog.message = "Model list loaded; its cache could not be saved."; + return catalog; + } catch { + const message = source === "ocx" ? "OCX model discovery failed. Check OCX and refresh." : "Codex model catalog is unavailable. Check its configured path and refresh."; + return cached ? { ...cached, status: "stale", message: `${message} Showing the last successful list.` } + : { state: "unavailable", entries: [], status: "unavailable", source, fetchedAt: null, message }; + } + }; + const request = query(); + pending.set(pendingKey, request); + try { return await request; } finally { pending.delete(pendingKey); } +} diff --git a/plugins/codexclaw/components/subagent-config/dist/mcp.js b/plugins/codexclaw/components/subagent-config/dist/mcp.js index 2c04a0e7..3369fb17 100644 --- a/plugins/codexclaw/components/subagent-config/dist/mcp.js +++ b/plugins/codexclaw/components/subagent-config/dist/mcp.js @@ -6,18 +6,17 @@ * - Persist per-role subagent config: default-model vs multi-model mapping, * and per-role prompt overrides (store: .codexclaw/subagents.json). * - Expose the config to the codexclaw GUI and as MCP tools. - * - Model catalog comes from the Codex config cache (CODEX_MODELS_CACHE_PATH). - * opencodex SYNCS its routed `provider/model` models into that codex config, so - * ocx-backed models appear in the catalog WITHOUT codexclaw calling ocx directly - * (detect-only boundary). `catalog_list` reads that cache via buildCatalog(). + * - Model catalog uses read-only OCX discovery and a shared CXC cache. + * When OCX is absent, the configured Codex catalog is read instead. * * Current scope: a spec-compliant stdio MCP server that completes the JSON-RPC * `initialize` handshake and advertises the subagent config/catalog tools below. * Zero third-party deps: newline-delimited JSON-RPC over stdin/stdout (node:* only). */ import { createInterface } from "node:readline"; -import { readConfig, setRole, ROLES, EFFORTS, } from "./store.js"; -import { buildCatalog } from "./catalog.js"; +import { ROLES, EFFORTS } from "./store.js"; +import { getSettings, updateSettings } from "./settings-api.js"; +import { readCatalog } from "./live-catalog.js"; const PROTOCOL_VERSION = "2024-11-05"; const SERVER_INFO = { name: "codexclaw-subagent-config", version: "0.1.1" }; @@ -33,8 +32,8 @@ function reply(id , result ) { const TOOLS = [ { name: "subagents_get", - description: "Read the per-role subagent config (explorer/reviewer/executor): mode, model, promptOverride.", - inputSchema: { type: "object", properties: {}, additionalProperties: false }, + description: "Read the per-role subagent config (explorer/reviewer/executor): mode, model, effort, promptOverride, source and scope. Project defaults to global, then original session.", + inputSchema: { type: "object", properties: { scope: { type: "string", enum: ["project", "global"] } }, additionalProperties: false }, }, { name: "subagents_set", @@ -43,6 +42,8 @@ const TOOLS = [ inputSchema: { type: "object", properties: { + scope: { type: "string", enum: ["project", "global"] }, + inherit: { type: "boolean", description: "Remove this entire role override and inherit the next scope; do not combine with role settings." }, role: { type: "string", enum: [...ROLES] }, mode: { type: "string", enum: ["default", "model"] }, model: { type: ["string", "null"] }, @@ -68,41 +69,27 @@ function toolError(id , message ) { reply(id, { content: [{ type: "text", text: JSON.stringify({ error: message }) }], isError: true }); } -function callTool(id , params ) { +async function callTool(id , params ) { const cwd = process.cwd(); const args = params.arguments ?? {}; - if (params.name === "subagents_get") { - toolResult(id, readConfig(cwd)); - return; - } - if (params.name === "subagents_set") { - const role = args.role ; - if (!ROLES.includes(role)) { - toolError(id, `unknown role "${String(args.role)}"`); - return; - } - const patch = {}; - if (args.mode !== undefined) patch.mode = args.mode; - if (args.model !== undefined) patch.model = args.model; - if (args.effort !== undefined) patch.effort = args.effort; - if (args.promptOverride !== undefined) patch.promptOverride = args.promptOverride; + if (params.name === "subagents_get" || params.name === "subagents_set") { try { - toolResult(id, setRole(cwd, role, patch)); + toolResult(id, params.name === "subagents_get" ? getSettings(cwd, args.scope) : updateSettings(cwd, args)); } catch (err) { toolError(id, err instanceof Error ? err.message : String(err)); } return; } if (params.name === "catalog_list") { - // Catalog read from the Codex config cache (native + ocx-synced routed slugs). - // L24 owns selection persistence; this never writes and never calls ocx directly. - toolResult(id, buildCatalog()); + // Read-only OCX discovery may update the shared catalog cache. + // Model and effort preferences are never changed by discovery. + toolResult(id, await readCatalog()); return; } toolError(id, `unknown tool: ${String(params.name)}`); } -function handle(msg ) { +async function handle(msg ) { const { id, method } = msg; switch (method) { case "initialize": @@ -116,7 +103,7 @@ function handle(msg ) { reply(id, { tools: TOOLS }); return; case "tools/call": - callTool(id, (msg ).params ?? {}); + await callTool(id, (msg ).params ?? {}); return; case "ping": reply(id, {}); @@ -134,7 +121,7 @@ rl.on("line", (line ) => { const trimmed = line.trim(); if (!trimmed) return; try { - handle(JSON.parse(trimmed) ); + void handle(JSON.parse(trimmed) ).catch(() => { /* malformed requests do not crash stdio */ }); } catch { // Malformed line: ignore rather than crash the long-lived server. } diff --git a/plugins/codexclaw/components/subagent-config/dist/settings-api.js b/plugins/codexclaw/components/subagent-config/dist/settings-api.js new file mode 100644 index 00000000..004cd64d --- /dev/null +++ b/plugins/codexclaw/components/subagent-config/dist/settings-api.js @@ -0,0 +1,31 @@ +/** Shared browser/MCP settings contract. Scope paths are host-owned, never request paths. */ +import { configScope, readSettings, resetRole, setRole, ROLES, } from './store.js'; + +export function getSettings(cwd , scope ) { + return readSettings(cwd, configScope(scope)); +} + +export function updateSettings(cwd , body ) { + if (!body || typeof body !== 'object' || Array.isArray(body)) throw new Error('missing body'); + const b = body ; + const role = b.role ; + if (!ROLES.includes(role)) throw new Error(`unknown role "${String(b.role)}"`); + const scope = configScope(b.scope); + if (b.inherit !== undefined && typeof b.inherit !== 'boolean') throw new Error('inherit must be a boolean'); + const patch = {}; + for (const key of ['mode', 'model', 'effort', 'promptOverride']) { + if (b[key] !== undefined) patch[key] = b[key]; + } + if (b.inherit === true) { + if (Object.keys(patch).length) throw new Error('inherit cannot be combined with role settings'); + resetRole(cwd, role, scope); + } else { + setRole(cwd, role, patch, scope); + } + return readSettings(cwd, scope); +} + +export function settingsResponse(operation ) { + try { return { status: 200, body: operation() }; } + catch (err) { return { status: 400, body: { error: err instanceof Error ? err.message : String(err) } }; } +} diff --git a/plugins/codexclaw/components/subagent-config/dist/store.js b/plugins/codexclaw/components/subagent-config/dist/store.js index ba46dfb5..fb86623d 100644 --- a/plugins/codexclaw/components/subagent-config/dist/store.js +++ b/plugins/codexclaw/components/subagent-config/dist/store.js @@ -4,13 +4,15 @@ * Per-role subagent model mode + prompt override for the three Phase-1 roles * (explorer/reviewer/executor). Missing file -> defaults; malformed values are * normalized per-field (strict reconstruct, never throws on read). Writes are - * atomic (temp + rename). NEVER mutates global Codex config; default mode needs + * atomic (temp + rename). User defaults live in CODEXCLAW_HOME; native + * Codex config is never mutated. Default mode needs * no ocx (uses the main Codex model). */ import { existsSync, mkdirSync, readFileSync, realpathSync, writeFileSync, rmSync } from "node:fs"; -import { join, resolve } from "node:path"; +import { dirname, join, resolve } from "node:path"; import { spawnSync } from "node:child_process"; -import { createHash } from "node:crypto"; +import { homedir } from "node:os"; +import { createHash, randomUUID } from "node:crypto"; import { renameWithRetry } from "./atomic-write.js"; export const STATE_DIR = ".codexclaw"; @@ -26,12 +28,8 @@ export const ROLES = ["explorer", "reviewer", "executor"] ; * the parent session's effort (the jawcode/cli-jaw policy default). An invalid * effort HARD-FAILS the spawn on the codex side, so the store validates on write. */ -// SCOPED to the universally-supported set: codex-rs validates the requested effort -// against the resolved model's supported_reasoning_levels and HARD-FAILS the spawn on -// a miss (multi_agents_common.rs validate_spawn_agent_reasoning_effort). Every model in -// the live catalog supports exactly {low,medium,high,xhigh}; the ReasoningEffort enum -// also defines none/minimal/max/ultra, but no selectable model advertises them, so -// offering them would let a saved config brick every later spawn. +// Supported wire values retained for backward compatibility. Model capabilities vary; +// the dashboard narrows these options using the current catalog's reasoningEfforts. export const EFFORTS = ["low", "medium", "high", "xhigh"] ; @@ -77,30 +75,98 @@ function reconstructRole(raw ) { return { mode, model, effort, promptOverride }; } -/** - * Read + normalize the config. Missing file -> defaults. Malformed JSON -> - * defaults (never throws). Each role is strictly reconstructed. - */ -export function readConfig(cwd ) { - const path = storePath(cwd); - if (!existsSync(path)) return defaultConfig(); - let parsed ; + + + + + + + + + +export function configScope(value = "project") { + if (value !== "project" && value !== "global") throw new Error(`invalid scope "${String(value)}"`); + return value; +} + +export function cxcHome(env = process.env) { + return env.CODEXCLAW_HOME?.trim() || join(homedir(), ".codexclaw"); +} + +export function globalStorePath(env = process.env) { + return join(cxcHome(env), STORE_FILE); +} + +/** Compatibility with the unpublished first scoped-settings patch. Reads never migrate. */ +function readGlobalRaw(env , forWrite = false) { + const canonical = globalStorePath(env); + if (!existsSync(canonical) && !env.CODEXCLAW_HOME?.trim()) { + const legacy = join(env.CODEX_HOME?.trim() || join(homedir(), ".codex"), "codexclaw", STORE_FILE); + if (existsSync(legacy)) return readRaw(legacy, forWrite); + } + return readRaw(canonical, forWrite); +} + +function scopedPath(cwd , scope , env ) { + return configScope(scope) === "global" ? globalStorePath(env) : storePath(cwd); +} + + +function readRaw(path , forWrite = false) { try { - parsed = JSON.parse(readFileSync(path, "utf8")); - } catch { - return defaultConfig(); + const parsed = JSON.parse(readFileSync(path, "utf8")); + if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) throw new Error("config must be an object"); + const raw = parsed ; + if (raw.roles !== undefined && (!raw.roles || typeof raw.roles !== "object" || Array.isArray(raw.roles))) { + throw new Error("roles must be an object"); + } + return { ...raw, roles: { ...(raw.roles ) } }; + } catch (err) { + if (forWrite && (err ).code !== "ENOENT") { + throw new Error(`cannot update subagent config: ${err instanceof Error ? err.message : String(err)}`); + } + return { roles: {} }; } - const roles = (parsed && typeof parsed === "object" ? (parsed ).roles : null) +} +function projectTrustWarning(cwd , env ) { + if (!isTrackedProjectConfig(cwd)) return undefined; + const token = projectConfigTrustToken(cwd); + if (token !== null && env.CODEXCLAW_TRUST_PROJECT_SUBAGENTS === token) return undefined; + return "ignored Git-tracked .codexclaw/subagents.json; review it, then run `cxc subagents trust-token` and export the printed project-bound value"; +} - ; - const out = defaultConfig(); - if (roles && typeof roles === "object") { - for (const role of ROLES) out.roles[role] = reconstructRole(roles[role]); +/** Resolve whole roles, preserving explicit null as original-session inheritance. */ +export function readSettings(cwd , scope = "project", env = process.env) { + configScope(scope); + const global = readGlobalRaw(env); + const project = scope === "project" ? readRaw(storePath(cwd)) : { roles: {} }; + const trustWarning = scope === "project" ? projectTrustWarning(cwd, env) : undefined; + const out = { + ...defaultConfig(), scope, + sources: { explorer: "session", reviewer: "session", executor: "session" }, + overrides: { explorer: false, reviewer: false, executor: false }, + ...(trustWarning ? { trustWarning } : {}), + }; + for (const role of ROLES) { + out.overrides[role] = Object.hasOwn((scope === "project" ? project : global).roles, role); + if (Object.hasOwn(global.roles, role)) { + out.roles[role] = reconstructRole(global.roles[role]); + out.sources[role] = "global"; + } + if (!trustWarning && Object.hasOwn(project.roles, role)) { + out.roles[role] = reconstructRole(project.roles[role]); + out.sources[role] = "project"; + } } return out; } +/** Effective config without UI metadata; reads never change persisted settings. */ +export function readConfig(cwd , scope = "project", env = process.env) { + return { roles: readSettings(cwd, scope, env).roles }; +} + /** Validate a role patch, returning an error message or null. */ export function validateRolePatch(patch ) { if (patch.mode !== undefined && patch.mode !== "default" && patch.mode !== "model") { @@ -122,42 +188,48 @@ export function validateRolePatch(patch ) { return null; } -/** Atomic write: temp file then rename. Creates .codexclaw/ if needed. */ -export function writeConfig(cwd , config ) { - const dir = join(cwd, STATE_DIR); - if (!existsSync(dir)) mkdirSync(dir, { recursive: true, mode: 0o700 }); - const path = storePath(cwd); - const tmp = `${path}.tmp`; +/** Atomic write with an exclusive temporary file; preserve unrelated JSON fields. */ +function writeRaw(path , config ) { + mkdirSync(dirname(path), { recursive: true, mode: 0o700 }); + const tmp = `${path}.${randomUUID()}.tmp`; try { - writeFileSync(tmp, `${JSON.stringify(config, null, 2)}\n`, { mode: 0o600 }); + writeFileSync(tmp, `${JSON.stringify(config, null, 2)}\n`, { mode: 0o600, flag: "wx" }); renameWithRetry(tmp, path); - } catch (err) { - try { - if (existsSync(tmp)) rmSync(tmp); - } catch { - // best-effort cleanup - } - throw err; + } finally { + rmSync(tmp, { force: true }); } } -/** - * Apply a patch to one role and persist. Returns the updated config. - * Validates the MERGED role (not the bare patch), so `{mode:"model"}` alone is - * valid when the role already has a saved model — the GUI checkbox depends on - * this. Throws on an invalid merged state (caller surfaces the message). - */ -export function setRole(cwd , role , patch ) { +/** Explicit full-config writes remain available to existing callers. */ +export function writeConfig(cwd , config ) { + writeRaw(storePath(cwd), config); +} + +/** Merge only the selected role; missing roles continue to inherit dynamically. */ +export function setRole(cwd , role , patch , scope = "project", env = process.env) { if (!ROLES.includes(role)) throw new Error(`unknown role "${role}"`); - const config = readConfig(cwd); - const next = { ...config.roles[role], ...patch }; + const path = scopedPath(cwd, scope, env); + const raw = scope === "global" ? readGlobalRaw(env, true) : readRaw(path, true); + const current = Object.hasOwn(raw.roles, role) ? reconstructRole(raw.roles[role]) : readConfig(cwd, scope, env).roles[role]; + const next = { ...current, ...patch }; const err = validateRolePatch(next); if (err) throw new Error(err); - // enforce the default-mode invariant: default mode ignores model. if (next.mode === "default") next.model = null; - config.roles[role] = next; - writeConfig(cwd, config); - return config; + raw.roles[role] = { ...(typeof raw.roles[role] === "object" && raw.roles[role] !== null ? raw.roles[role] : {}), ...next }; + writeRaw(path, raw); + return readConfig(cwd, scope, env); +} + +/** Remove a role override. null fields deliberately do not perform this action. */ +export function resetRole(cwd , role , scope = "project", env = process.env) { + if (!ROLES.includes(role)) throw new Error(`unknown role "${role}"`); + const path = scopedPath(cwd, scope, env); + const raw = scope === "global" ? readGlobalRaw(env, true) : readRaw(path, true); + if (Object.hasOwn(raw.roles, role)) { + delete raw.roles[role]; + writeRaw(path, raw); + } + return readConfig(cwd, scope, env); } @@ -218,20 +290,8 @@ export function resolveSpawnConfig( role , env = process.env, ) { - const tracked = isTrackedProjectConfig(cwd); - const expectedTrust = tracked ? projectConfigTrustToken(cwd) : null; - if (tracked && (expectedTrust === null || env.CODEXCLAW_TRUST_PROJECT_SUBAGENTS !== expectedTrust)) { - return { - role, - model: null, - usesMainModel: true, - effort: null, - promptOverride: null, - trustWarning: - "ignored Git-tracked .codexclaw/subagents.json; review it, then run `cxc subagents trust-token` and export the printed project-bound value", - }; - } - const cfg = readConfig(cwd).roles[role]; + const settings = readSettings(cwd, "project", env); + const cfg = settings.roles[role]; const usesMainModel = cfg.mode === "default"; return { role, @@ -239,5 +299,6 @@ export function resolveSpawnConfig( usesMainModel, effort: cfg.effort, promptOverride: cfg.promptOverride, + ...(settings.trustWarning ? { trustWarning: settings.trustWarning } : {}), }; } diff --git a/plugins/codexclaw/components/subagent-config/dist/win-exec.js b/plugins/codexclaw/components/subagent-config/dist/win-exec.js new file mode 100644 index 00000000..2e778e07 --- /dev/null +++ b/plugins/codexclaw/components/subagent-config/dist/win-exec.js @@ -0,0 +1,89 @@ +/** + * win-exec.ts - one entry point for spawning external commands. + * + * Three Windows facts drive this: + * 1. npm-installed CLIs are `.cmd` shims, and Node refuses shell-less `.cmd` spawns + * after the CVE-2024-27980 hardening. + * 2. A bare command name skips PATHEXT resolution, so `spawn("npm")` ENOENTs even + * with `npm.cmd` on PATH (measured, 002 B4). + * 3. `shell: true` is not a fix: Node does not escape cmd metacharacters there, so + * a path containing `&` or `^` becomes a command injection. + * + * Env vars are read case-insensitively: a spawned child can arrive with `Path`, + * `PATH`, or both, and reading one fixed spelling resolves against the wrong list + * (001 3.2). + */ +import { existsSync } from "node:fs"; +import { isAbsolute, join } from "node:path"; + + + + + + + +/** Case-insensitive env lookup with a key-scan fallback. */ +export function envValue(env , name ) { + const direct = env[name]; + if (direct !== undefined) return direct; + const lower = name.toLowerCase(); + for (const key of Object.keys(env)) { + if (key.toLowerCase() === lower) return env[key]; + } + return undefined; +} + +const DEFAULT_PATHEXT = ".COM;.EXE;.BAT;.CMD"; + +/** Resolve `command` against PATH + PATHEXT. Returns the input when nothing matches. */ +export function resolveWindowsCommand(command , env ) { + if (command.includes("/") || command.includes("\\") || isAbsolute(command)) return command; + const exts = (envValue(env, "PATHEXT") ?? DEFAULT_PATHEXT).split(";").filter((e) => e.length > 0); + // A WINDOWS PATH is always ";"-separated. node:path's `delimiter` follows the + // HOST, so on a Linux runner exercising this win32-only walk it would be ":" + // and the whole PATH would collapse into one bogus directory entry. + const dirs = (envValue(env, "PATH") ?? "").split(";").filter((d) => d.length > 0); + for (const dir of dirs) { + for (const ext of exts) { + const candidate = join(dir, command + ext); + if (existsSync(candidate)) return candidate; + // Case-sensitive filesystems (WSL, Linux CI): PATHEXT spells ".EXE" but + // real shims are "npm.cmd" / "gh.exe". Retry the lowercased extension. + const lowered = join(dir, command + ext.toLowerCase()); + if (existsSync(lowered)) return lowered; + } + } + return command; +} + +/** cross-spawn's escaping: double backslashes before quotes, quote, then caret. */ +function escapeCmdArg(arg ) { + let out = arg.replace(/(\\*)"/g, '$1$1\\"').replace(/(\\*)$/, "$1$1"); + out = `"${out}"`; + return out.replace(/[()%!^"<>&|;, ]/g, "^$&"); +} + +function escapeCmdCommand(command ) { + return command.replace(/[()%!^"<>&|;, ]/g, "^$&"); +} + +/** + * Build the spawn shape for `command`. POSIX is a passthrough; win32 resolves + * PATHEXT and routes only `.cmd`/`.bat` through cmd.exe. + */ +export function commandInvocation( + command , + args , + platform = process.platform, + env = process.env, +) { + if (platform !== "win32") return { file: command, args: [...args], options: {} }; + const resolved = resolveWindowsCommand(command, env); + if (!/\.(cmd|bat)$/i.test(resolved)) return { file: resolved, args: [...args], options: {} }; + const line = [escapeCmdCommand(resolved), ...args.map(escapeCmdArg)].join(" "); + return { + file: envValue(env, "ComSpec") ?? "cmd.exe", + args: ["/d", "/s", "/c", `"${line}"`], + options: { windowsVerbatimArguments: true }, + }; +} diff --git a/plugins/codexclaw/components/subagent-config/src/catalog.ts b/plugins/codexclaw/components/subagent-config/src/catalog.ts index e396efa1..647a1dbd 100644 --- a/plugins/codexclaw/components/subagent-config/src/catalog.ts +++ b/plugins/codexclaw/components/subagent-config/src/catalog.ts @@ -1,24 +1,7 @@ -/** - * catalog.ts — selectable model catalog (L25 / 250-252). - * - * Source = Codex-native catalog (always) + ocx-backed models (when ocx is - * detected and exposes a catalog). Native entries come first; entries are - * deduplicated by stable model id keeping the native one. No network fetch, no - * vendored ocx files, no selected-model persistence (L24 owns that). - * - * Native source: the Codex live catalog cache at CODEX_MODELS_CACHE_PATH, read - * through an allowlist. When the cache is absent/unreadable, fall back to the - * documented NATIVE_OPENAI_MODELS set (opencodex src/codex-catalog.ts:44). - * - * Slug parity (L9.2 / 092): the LIVE Codex catalog keys each entry by `slug` - * (bare like "gpt-5.5", or routed "provider/model"), not `id` (opencodex - * codex-catalog.ts:152,183). The cache reader therefore accepts BOTH `id` and - * `slug`, and dedup compares on the resolved key so a native slug and an ocx id - * for the same model collapse, native kept first. - */ +/** Native catalog parsing and pure catalog composition. Live OCX discovery lives in live-catalog.ts. */ import { existsSync, readFileSync } from "node:fs"; import { homedir } from "node:os"; -import { join } from "node:path"; +import { isAbsolute, join, resolve } from "node:path"; export const NATIVE_OPENAI_MODELS = ["gpt-5.5", "gpt-5.4", "gpt-5.4-mini", "gpt-5.6-luna"] as const; @@ -26,11 +9,12 @@ export type ModelSource = "native" | "ocx"; export interface CatalogEntry { id: string; + reasoningEfforts?: string[] | null; source: ModelSource; label: string; } -export type CatalogState = "native-catalog" | "ocx-active" | "unsupported-ocx-catalog"; +export type CatalogState = "native-catalog" | "ocx-active" | "unsupported-ocx-catalog" | "unavailable"; export interface Catalog { state: CatalogState; @@ -71,47 +55,62 @@ function isRoutedSlug(key: string): boolean { return key.includes("/"); } -/** Read the Codex live catalog cache (CODEX_MODELS_CACHE_PATH) through the - * allowlist. Reads each entry by `id` OR `slug` (live catalog uses slug). - * Returns ids or null when absent/unreadable. - * - * L20/WP4: the cache is the codex config catalog, which opencodex SYNCS its - * routed `provider/model` slugs into. codexclaw reads that config (it never - * calls ocx directly). So the allowlist admits BOTH the documented native ids - * AND any routed slug (contains "/") — dropping routed slugs would hide exactly - * the ocx-synced models the subagent config is meant to select. */ -export function readNativeCacheDefault(env: NodeJS.ProcessEnv = process.env): string[] | null { - // Resolve like opencodex (codex-paths.ts:30): explicit override, else - // $CODEX_HOME/models_cache.json, else ~/.codex/models_cache.json. Nothing in - // `cxc serve` sets CODEX_MODELS_CACHE_PATH, so the homedir default is what - // makes the ocx-synced routed slugs actually load in practice. - const path = - env.CODEX_MODELS_CACHE_PATH ?? - join(env.CODEX_HOME ?? join(homedir(), ".codex"), "models_cache.json"); - if (!existsSync(path)) return null; +/** Root-level TOML path only: never read a similarly named key inside a table. */ +export function nativeCatalogPath(env: NodeJS.ProcessEnv = process.env): string | null { + if (env.CODEX_MODELS_CACHE_PATH?.trim()) return env.CODEX_MODELS_CACHE_PATH; + const home = env.CODEX_HOME?.trim() || join(homedir(), ".codex"); try { - const parsed = JSON.parse(readFileSync(path, "utf8")) as unknown; - const list = Array.isArray(parsed) ? parsed : (parsed as { models?: unknown })?.models; + const lines = readFileSync(join(home, "config.toml"), "utf8").split(/\r?\n/); + for (const line of lines) { + if (/^\s*\[/.test(line)) break; + if (!/^\s*(?:model_catalog_json|"model_catalog_json"|'model_catalog_json')\s*=/.test(line)) continue; + const match = /^\s*(?:model_catalog_json|"model_catalog_json"|'model_catalog_json')\s*=\s*("(?:\\.|[^"\\])*"|'[^']*')\s*(?:#.*)?$/.exec(line); + if (!match) return null; + const value: string = match[1].startsWith("'") ? match[1].slice(1, -1) : JSON.parse(match[1]); + if (!value.trim()) return null; + const expanded = value.startsWith("~/") || value.startsWith("~\\") ? join(homedir(), value.slice(2)) : value; + return isAbsolute(expanded) ? expanded : resolve(home, expanded); + } + } catch (error) { + if ((error as NodeJS.ErrnoException).code !== "ENOENT") return null; + } + return join(home, "models_cache.json"); +} + +export function reasoningEfforts(raw: unknown): string[] | null { + if (!Array.isArray(raw)) return null; + return [...new Set(raw.flatMap(value => { + const effort = typeof value === "string" ? value : value && typeof value === "object" ? (value as { effort?: unknown }).effort : undefined; + return typeof effort === "string" && effort.length > 0 ? [effort] : []; + }))]; +} + +export function readNativeCatalog(env: NodeJS.ProcessEnv = process.env): CatalogEntry[] | null { + const path = nativeCatalogPath(env); + if (!path || !existsSync(path)) return null; + try { + const parsed: unknown = JSON.parse(readFileSync(path, "utf8")); + const list = Array.isArray(parsed) ? parsed : (parsed as { models?: unknown } | null)?.models; if (!Array.isArray(list)) return null; - const ids = list.map(entryKey).filter((x): x is string => typeof x === "string"); - // allowlist: ship documented native ids AND routed provider/model slugs - // (the ocx-synced entries). Dedup preserves first-seen order so a slug+id - // duplicate yields one entry. const seen = new Set(); - const allowed = ids.filter( - (id) => - ((NATIVE_OPENAI_MODELS as readonly string[]).includes(id) || isRoutedSlug(id)) && - !seen.has(id) && - (seen.add(id), true), - ); - return allowed.length ? allowed : null; - } catch { - return null; - } + return list.flatMap(raw => { + const id = entryKey(raw); + const row = raw && typeof raw === "object" ? raw as Record : {}; + if (!id?.trim() || seen.has(id) || row.disabled === true || row.visibility === "hide") return []; + seen.add(id); + const source: ModelSource = isRoutedSlug(id) ? "ocx" : "native"; + return [{ id, source, label: id, reasoningEfforts: reasoningEfforts(row.reasoningEfforts ?? row.supported_reasoning_levels) }]; + }); + } catch { return null; } +} + +/** Legacy ID-only reader retained for pure consumers and fixtures. */ +export function readNativeCacheDefault(env: NodeJS.ProcessEnv = process.env): string[] | null { + return readNativeCatalog(env)?.map(entry => entry.id) ?? null; } function nativeEntries(deps: CatalogDeps): CatalogEntry[] { - const ids = (deps.readNativeCache ?? readNativeCacheDefault)() ?? [...NATIVE_OPENAI_MODELS]; + const ids = (deps.readNativeCache ?? readNativeCacheDefault)() ?? []; // Entries from the codex config cache: bare ids are native; routed `provider/model` // slugs were synced in by opencodex, so label them as ocx-origin even though they // arrive through the native cache (codexclaw never calls ocx directly). @@ -131,7 +130,7 @@ export function buildCatalog(deps: CatalogDeps = {}): Catalog { const status = deps.providerStatus; if (!status || status.mode !== "provider") { - return { state: "native-catalog", entries: native }; + return { state: native.length ? "native-catalog" : "unavailable", entries: native }; } // ocx is active. If it exposes no catalog interface, the cache-sync channel may diff --git a/plugins/codexclaw/components/subagent-config/src/cli.ts b/plugins/codexclaw/components/subagent-config/src/cli.ts index 806a2a38..3caebd2f 100644 --- a/plugins/codexclaw/components/subagent-config/src/cli.ts +++ b/plugins/codexclaw/components/subagent-config/src/cli.ts @@ -13,13 +13,14 @@ * subagents set --mode default|model [--model ] [--effort |--clear-effort] * [--prompt |--clear-prompt] */ -import { readConfig, setRole, projectConfigTrustToken, ROLES, EFFORTS, type RoleName, type RoleConfig, type EffortName } from "./store.ts"; +import { readConfig, setRole, resetRole, type ConfigScope, projectConfigTrustToken, ROLES, EFFORTS, type RoleName, type RoleConfig, type EffortName } from "./store.ts"; import { realpathSync } from "node:fs"; import { fileURLToPath } from "node:url"; export interface ParsedSubagentsArgs { - action: "list" | "get" | "set" | "trust-token" | "help"; + action: "list" | "get" | "set" | "reset" | "trust-token" | "help"; role?: RoleName; + scope?: ConfigScope; patch?: Partial; error?: string; } @@ -30,11 +31,23 @@ function isRole(v: string | undefined): v is RoleName { /** Pure structural parse of the `subagents` argv (excluding the leading verb). */ export function parseSubagentsArgs(argv: string[]): ParsedSubagentsArgs { + // Scope is an explicit trailing selector, so prompt/model values stay literal. + if (argv.at(-1) === "--global" && !["--prompt", "--model"].includes(argv.at(-2) ?? "")) { + return { ...parseProjectArgs(argv.slice(0, -1)), scope: "global" }; + } + return parseProjectArgs(argv); +} + +function parseProjectArgs(argv: string[]): ParsedSubagentsArgs { const sub = argv[0]; if (sub === undefined || sub === "list") return { action: "list" }; if (sub === "help" || sub === "--help" || sub === "-h") return { action: "help" }; if (sub === "trust-token") return { action: "trust-token" }; + if (sub === "reset") { + if (!isRole(argv[1]) || argv.length !== 2) return { action: "reset", error: "reset requires exactly one valid role" }; + return { action: "reset", role: argv[1] }; + } if (sub === "get") { if (!isRole(argv[1])) return { action: "get", error: `unknown role '${argv[1] ?? ""}' (expected ${ROLES.join("|")})` }; return { action: "get", role: argv[1] }; @@ -83,6 +96,8 @@ const HELP = [ " subagents list all role configs", " subagents get show one role config", " subagents set --mode default|model [--model ] [--effort |--clear-effort] [--prompt |--clear-prompt]", + " subagents reset remove the role override and inherit the next scope", + " Append --global to list/get/set/reset to manage user defaults", " subagents trust-token print an export bound to this repo and exact config", "", ` roles: ${ROLES.join(", ")}`, @@ -101,9 +116,9 @@ export function runSubagents(parsed: ParsedSubagentsArgs, cwd: string): Subagent case "help": return { code: 0, output: HELP }; case "list": - return { code: 0, output: JSON.stringify(readConfig(cwd), null, 2) }; + return { code: 0, output: JSON.stringify(readConfig(cwd, parsed.scope), null, 2) }; case "get": { - const cfg = readConfig(cwd); + const cfg = readConfig(cwd, parsed.scope); return { code: 0, output: JSON.stringify(cfg.roles[parsed.role as RoleName], null, 2) }; } case "trust-token": { @@ -111,9 +126,17 @@ export function runSubagents(parsed: ParsedSubagentsArgs, cwd: string): Subagent if (!token) return { code: 1, output: "subagents: cannot hash .codexclaw/subagents.json" }; return { code: 0, output: `export CODEXCLAW_TRUST_PROJECT_SUBAGENTS='${token}'` }; } + case "reset": { + try { + const cfg = resetRole(cwd, parsed.role as RoleName, parsed.scope); + return { code: 0, output: JSON.stringify(cfg.roles[parsed.role as RoleName], null, 2) }; + } catch (err) { + return { code: 1, output: `subagents: ${err instanceof Error ? err.message : String(err)}` }; + } + } case "set": { try { - const cfg = setRole(cwd, parsed.role as RoleName, parsed.patch ?? {}); + const cfg = setRole(cwd, parsed.role as RoleName, parsed.patch ?? {}, parsed.scope); return { code: 0, output: JSON.stringify(cfg.roles[parsed.role as RoleName], null, 2) }; } catch (err) { return { code: 1, output: `subagents: ${err instanceof Error ? err.message : String(err)}` }; diff --git a/plugins/codexclaw/components/subagent-config/src/live-catalog.ts b/plugins/codexclaw/components/subagent-config/src/live-catalog.ts new file mode 100644 index 00000000..ae62c9a7 --- /dev/null +++ b/plugins/codexclaw/components/subagent-config/src/live-catalog.ts @@ -0,0 +1,119 @@ +/** Read-only OCX discovery with a shared, bounded-age last-success cache. */ +import { execFile } from "node:child_process"; +import { createHash, randomUUID } from "node:crypto"; +import { mkdirSync, readFileSync, writeFileSync, rmSync } from "node:fs"; +import { join } from "node:path"; +import { readNativeCatalog, reasoningEfforts, type Catalog, type CatalogEntry } from "./catalog.ts"; +import { cxcHome } from "./store.ts"; +import { commandInvocation } from "./win-exec.ts"; +import { renameWithRetry } from "./atomic-write.ts"; + +export const CATALOG_TTL_MS = 30_000; +export interface LiveCatalog extends Catalog { + status: "fresh" | "stale" | "unavailable"; + source: "ocx" | "native"; + fetchedAt: string | null; + message?: string; +} +export interface CatalogOptions { + forceRefresh?: boolean; + env?: NodeJS.ProcessEnv; + now?: () => number; + runOcx?: (env: NodeJS.ProcessEnv) => Promise; + readNative?: (env: NodeJS.ProcessEnv) => CatalogEntry[] | null; +} +const pending = new Map>(); + +export function runOcxModels(env: NodeJS.ProcessEnv): Promise { + const invocation = commandInvocation("ocx", ["models", "live", "--json"], process.platform, env); + return new Promise((resolve, reject) => { + execFile(invocation.file, invocation.args, { + ...invocation.options, env, encoding: "utf8", timeout: 12_000, + killSignal: "SIGKILL", maxBuffer: 4 * 1024 * 1024, windowsHide: true, + }, (error, stdout) => error ? reject(error) : resolve(stdout)); + }); +} + +export function parseOcxModels(stdout: string): CatalogEntry[] { + const parsed: unknown = JSON.parse(stdout); + if (!Array.isArray(parsed)) throw new Error("invalid OCX catalog"); + const seen = new Set(); + return parsed.flatMap(raw => { + if (!raw || typeof raw !== "object" || Array.isArray(raw)) throw new Error("invalid OCX model row"); + const row = raw as Record; + if (row.disabled === true || row.initialSelectionPending === true) return []; + const id = typeof row.namespaced === "string" ? row.namespaced : row.native === true ? row.id : undefined; + if (typeof id !== "string" || !id.trim()) throw new Error("invalid OCX model id"); + if (seen.has(id)) return []; + seen.add(id); + return [{ id, source: row.native === true ? "native" as const : "ocx" as const, + label: typeof row.displayName === "string" && row.displayName.trim() ? `${row.displayName} (${id})` : id, + reasoningEfforts: reasoningEfforts(row.reasoningEfforts) }]; + }); +} + +function sourceKey(env: NodeJS.ProcessEnv): string { + return createHash("sha256").update(JSON.stringify([ + env.CODEX_HOME ?? "", env.CODEX_MODELS_CACHE_PATH ?? "", env.PATH ?? env.Path ?? "", env.OPENCODEX_HOME ?? "", + ])).digest("hex"); +} +function cachedCatalog(path: string, key: string, now: number): LiveCatalog | null { + try { + const raw = JSON.parse(readFileSync(path, "utf8")); + const c = raw.catalog as LiveCatalog; + if (raw.key !== key || !c || c.status !== "fresh" || !["ocx", "native"].includes(c.source) || !Array.isArray(c.entries)) return null; + if (!c.fetchedAt || !Number.isFinite(Date.parse(c.fetchedAt)) || Date.parse(c.fetchedAt) > now + 1000) return null; + if (!c.entries.every(entry => entry && typeof entry.id === "string" && entry.id.length && typeof entry.label === "string" + && ["ocx", "native"].includes(entry.source) && (entry.reasoningEfforts === null || Array.isArray(entry.reasoningEfforts) && entry.reasoningEfforts.every(e => typeof e === "string")))) return null; + return c; + } catch { return null; } +} +function persist(path: string, key: string, catalog: LiveCatalog, home: string): boolean { + const temp = `${path}.${randomUUID()}.tmp`; + try { + mkdirSync(home, { recursive: true, mode: 0o700 }); + writeFileSync(temp, JSON.stringify({ key, catalog }) + "\n", { flag: "wx", mode: 0o600 }); + renameWithRetry(temp, path); + return true; + } catch { return false; } + finally { try { rmSync(temp, { force: true }); } catch { /* best-effort own temp cleanup */ } } +} + +/** Project cwd never participates in cache identity. Explicit CXC home isolates all state. */ +export async function readCatalog(options: CatalogOptions = {}): Promise { + const env = options.env ?? process.env; + const now = options.now ?? Date.now; + const home = cxcHome(env); + const path = join(home, "model-catalog.json"); + const key = sourceKey(env); + const pendingKey = `${path}:${key}`; + const running = pending.get(pendingKey); + if (running) return running; + const cached = cachedCatalog(path, key, now()); + if (!options.forceRefresh && cached && now() - Date.parse(cached.fetchedAt!) < CATALOG_TTL_MS) return cached; + const query = async (): Promise => { + let source: "ocx" | "native" = "ocx"; + try { + let entries: CatalogEntry[]; + try { entries = parseOcxModels(await (options.runOcx ?? runOcxModels)(env)); } + catch (error) { + if ((error as NodeJS.ErrnoException).code !== "ENOENT") throw error; + source = "native"; + const native = (options.readNative ?? readNativeCatalog)(env); + if (native === null) throw new Error("native catalog unavailable"); + entries = native; + } + const catalog: LiveCatalog = { state: source === "ocx" ? "ocx-active" : "native-catalog", entries, + status: "fresh", source, fetchedAt: new Date(now()).toISOString() }; + if (!persist(path, key, catalog, home)) catalog.message = "Model list loaded; its cache could not be saved."; + return catalog; + } catch { + const message = source === "ocx" ? "OCX model discovery failed. Check OCX and refresh." : "Codex model catalog is unavailable. Check its configured path and refresh."; + return cached ? { ...cached, status: "stale", message: `${message} Showing the last successful list.` } + : { state: "unavailable", entries: [], status: "unavailable", source, fetchedAt: null, message }; + } + }; + const request = query(); + pending.set(pendingKey, request); + try { return await request; } finally { pending.delete(pendingKey); } +} diff --git a/plugins/codexclaw/components/subagent-config/src/mcp.ts b/plugins/codexclaw/components/subagent-config/src/mcp.ts index 0ad494fe..b44a3985 100644 --- a/plugins/codexclaw/components/subagent-config/src/mcp.ts +++ b/plugins/codexclaw/components/subagent-config/src/mcp.ts @@ -6,18 +6,17 @@ * - Persist per-role subagent config: default-model vs multi-model mapping, * and per-role prompt overrides (store: .codexclaw/subagents.json). * - Expose the config to the codexclaw GUI and as MCP tools. - * - Model catalog comes from the Codex config cache (CODEX_MODELS_CACHE_PATH). - * opencodex SYNCS its routed `provider/model` models into that codex config, so - * ocx-backed models appear in the catalog WITHOUT codexclaw calling ocx directly - * (detect-only boundary). `catalog_list` reads that cache via buildCatalog(). + * - Model catalog uses read-only OCX discovery and a shared CXC cache. + * When OCX is absent, the configured Codex catalog is read instead. * * Current scope: a spec-compliant stdio MCP server that completes the JSON-RPC * `initialize` handshake and advertises the subagent config/catalog tools below. * Zero third-party deps: newline-delimited JSON-RPC over stdin/stdout (node:* only). */ import { createInterface } from "node:readline"; -import { readConfig, setRole, ROLES, EFFORTS, type RoleName } from "./store.ts"; -import { buildCatalog } from "./catalog.ts"; +import { ROLES, EFFORTS } from "./store.ts"; +import { getSettings, updateSettings } from "./settings-api.ts"; +import { readCatalog } from "./live-catalog.ts"; const PROTOCOL_VERSION = "2024-11-05"; const SERVER_INFO = { name: "codexclaw-subagent-config", version: "0.1.1" }; @@ -33,8 +32,8 @@ function reply(id: unknown, result: unknown): void { const TOOLS = [ { name: "subagents_get", - description: "Read the per-role subagent config (explorer/reviewer/executor): mode, model, promptOverride.", - inputSchema: { type: "object", properties: {}, additionalProperties: false }, + description: "Read the per-role subagent config (explorer/reviewer/executor): mode, model, effort, promptOverride, source and scope. Project defaults to global, then original session.", + inputSchema: { type: "object", properties: { scope: { type: "string", enum: ["project", "global"] } }, additionalProperties: false }, }, { name: "subagents_set", @@ -43,6 +42,8 @@ const TOOLS = [ inputSchema: { type: "object", properties: { + scope: { type: "string", enum: ["project", "global"] }, + inherit: { type: "boolean", description: "Remove this entire role override and inherit the next scope; do not combine with role settings." }, role: { type: "string", enum: [...ROLES] }, mode: { type: "string", enum: ["default", "model"] }, model: { type: ["string", "null"] }, @@ -68,41 +69,27 @@ function toolError(id: unknown, message: string): void { reply(id, { content: [{ type: "text", text: JSON.stringify({ error: message }) }], isError: true }); } -function callTool(id: unknown, params: { name?: string; arguments?: Record }): void { +async function callTool(id: unknown, params: { name?: string; arguments?: Record }): Promise { const cwd = process.cwd(); const args = params.arguments ?? {}; - if (params.name === "subagents_get") { - toolResult(id, readConfig(cwd)); - return; - } - if (params.name === "subagents_set") { - const role = args.role as RoleName; - if (!ROLES.includes(role)) { - toolError(id, `unknown role "${String(args.role)}"`); - return; - } - const patch: Record = {}; - if (args.mode !== undefined) patch.mode = args.mode; - if (args.model !== undefined) patch.model = args.model; - if (args.effort !== undefined) patch.effort = args.effort; - if (args.promptOverride !== undefined) patch.promptOverride = args.promptOverride; + if (params.name === "subagents_get" || params.name === "subagents_set") { try { - toolResult(id, setRole(cwd, role, patch)); + toolResult(id, params.name === "subagents_get" ? getSettings(cwd, args.scope) : updateSettings(cwd, args)); } catch (err) { toolError(id, err instanceof Error ? err.message : String(err)); } return; } if (params.name === "catalog_list") { - // Catalog read from the Codex config cache (native + ocx-synced routed slugs). - // L24 owns selection persistence; this never writes and never calls ocx directly. - toolResult(id, buildCatalog()); + // Read-only OCX discovery may update the shared catalog cache. + // Model and effort preferences are never changed by discovery. + toolResult(id, await readCatalog()); return; } toolError(id, `unknown tool: ${String(params.name)}`); } -function handle(msg: { id?: unknown; method?: string }): void { +async function handle(msg: { id?: unknown; method?: string }): Promise { const { id, method } = msg; switch (method) { case "initialize": @@ -116,7 +103,7 @@ function handle(msg: { id?: unknown; method?: string }): void { reply(id, { tools: TOOLS }); return; case "tools/call": - callTool(id, (msg as { params?: { name?: string; arguments?: Record } }).params ?? {}); + await callTool(id, (msg as { params?: { name?: string; arguments?: Record } }).params ?? {}); return; case "ping": reply(id, {}); @@ -134,7 +121,7 @@ rl.on("line", (line: string) => { const trimmed = line.trim(); if (!trimmed) return; try { - handle(JSON.parse(trimmed) as { id?: unknown; method?: string }); + void handle(JSON.parse(trimmed) as { id?: unknown; method?: string }).catch(() => { /* malformed requests do not crash stdio */ }); } catch { // Malformed line: ignore rather than crash the long-lived server. } diff --git a/plugins/codexclaw/components/subagent-config/src/settings-api.ts b/plugins/codexclaw/components/subagent-config/src/settings-api.ts new file mode 100644 index 00000000..3229eed4 --- /dev/null +++ b/plugins/codexclaw/components/subagent-config/src/settings-api.ts @@ -0,0 +1,31 @@ +/** Shared browser/MCP settings contract. Scope paths are host-owned, never request paths. */ +import { configScope, readSettings, resetRole, setRole, ROLES, type RoleName, type SubagentSettings } from './store.ts'; + +export function getSettings(cwd: string, scope?: unknown): SubagentSettings { + return readSettings(cwd, configScope(scope)); +} + +export function updateSettings(cwd: string, body: unknown): SubagentSettings { + if (!body || typeof body !== 'object' || Array.isArray(body)) throw new Error('missing body'); + const b = body as Record; + const role = b.role as RoleName; + if (!ROLES.includes(role)) throw new Error(`unknown role "${String(b.role)}"`); + const scope = configScope(b.scope); + if (b.inherit !== undefined && typeof b.inherit !== 'boolean') throw new Error('inherit must be a boolean'); + const patch: Record = {}; + for (const key of ['mode', 'model', 'effort', 'promptOverride']) { + if (b[key] !== undefined) patch[key] = b[key]; + } + if (b.inherit === true) { + if (Object.keys(patch).length) throw new Error('inherit cannot be combined with role settings'); + resetRole(cwd, role, scope); + } else { + setRole(cwd, role, patch, scope); + } + return readSettings(cwd, scope); +} + +export function settingsResponse(operation: () => SubagentSettings): { status: number; body: unknown } { + try { return { status: 200, body: operation() }; } + catch (err) { return { status: 400, body: { error: err instanceof Error ? err.message : String(err) } }; } +} diff --git a/plugins/codexclaw/components/subagent-config/src/store.ts b/plugins/codexclaw/components/subagent-config/src/store.ts index ad31cc1e..1debd0e4 100644 --- a/plugins/codexclaw/components/subagent-config/src/store.ts +++ b/plugins/codexclaw/components/subagent-config/src/store.ts @@ -4,13 +4,15 @@ * Per-role subagent model mode + prompt override for the three Phase-1 roles * (explorer/reviewer/executor). Missing file -> defaults; malformed values are * normalized per-field (strict reconstruct, never throws on read). Writes are - * atomic (temp + rename). NEVER mutates global Codex config; default mode needs + * atomic (temp + rename). User defaults live in CODEXCLAW_HOME; native + * Codex config is never mutated. Default mode needs * no ocx (uses the main Codex model). */ import { existsSync, mkdirSync, readFileSync, realpathSync, writeFileSync, rmSync } from "node:fs"; -import { join, resolve } from "node:path"; +import { dirname, join, resolve } from "node:path"; import { spawnSync } from "node:child_process"; -import { createHash } from "node:crypto"; +import { homedir } from "node:os"; +import { createHash, randomUUID } from "node:crypto"; import { renameWithRetry } from "./atomic-write.ts"; export const STATE_DIR = ".codexclaw"; @@ -26,12 +28,8 @@ export type RoleMode = "default" | "model"; * the parent session's effort (the jawcode/cli-jaw policy default). An invalid * effort HARD-FAILS the spawn on the codex side, so the store validates on write. */ -// SCOPED to the universally-supported set: codex-rs validates the requested effort -// against the resolved model's supported_reasoning_levels and HARD-FAILS the spawn on -// a miss (multi_agents_common.rs validate_spawn_agent_reasoning_effort). Every model in -// the live catalog supports exactly {low,medium,high,xhigh}; the ReasoningEffort enum -// also defines none/minimal/max/ultra, but no selectable model advertises them, so -// offering them would let a saved config brick every later spawn. +// Supported wire values retained for backward compatibility. Model capabilities vary; +// the dashboard narrows these options using the current catalog's reasoningEfforts. export const EFFORTS = ["low", "medium", "high", "xhigh"] as const; export type EffortName = (typeof EFFORTS)[number]; @@ -77,30 +75,98 @@ function reconstructRole(raw: unknown): RoleConfig { return { mode, model, effort, promptOverride }; } -/** - * Read + normalize the config. Missing file -> defaults. Malformed JSON -> - * defaults (never throws). Each role is strictly reconstructed. - */ -export function readConfig(cwd: string): SubagentsConfig { - const path = storePath(cwd); - if (!existsSync(path)) return defaultConfig(); - let parsed: unknown; +export type ConfigScope = "project" | "global"; +export type ConfigSource = ConfigScope | "session"; +export interface SubagentSettings extends SubagentsConfig { + scope: ConfigScope; + sources: Record; + overrides: Record; + trustWarning?: string; +} + +export function configScope(value: unknown = "project"): ConfigScope { + if (value !== "project" && value !== "global") throw new Error(`invalid scope "${String(value)}"`); + return value; +} + +export function cxcHome(env: NodeJS.ProcessEnv = process.env): string { + return env.CODEXCLAW_HOME?.trim() || join(homedir(), ".codexclaw"); +} + +export function globalStorePath(env: NodeJS.ProcessEnv = process.env): string { + return join(cxcHome(env), STORE_FILE); +} + +/** Compatibility with the unpublished first scoped-settings patch. Reads never migrate. */ +function readGlobalRaw(env: NodeJS.ProcessEnv, forWrite = false): RawConfig { + const canonical = globalStorePath(env); + if (!existsSync(canonical) && !env.CODEXCLAW_HOME?.trim()) { + const legacy = join(env.CODEX_HOME?.trim() || join(homedir(), ".codex"), "codexclaw", STORE_FILE); + if (existsSync(legacy)) return readRaw(legacy, forWrite); + } + return readRaw(canonical, forWrite); +} + +function scopedPath(cwd: string, scope: ConfigScope, env: NodeJS.ProcessEnv): string { + return configScope(scope) === "global" ? globalStorePath(env) : storePath(cwd); +} + +type RawConfig = Record & { roles: Record }; +function readRaw(path: string, forWrite = false): RawConfig { try { - parsed = JSON.parse(readFileSync(path, "utf8")); - } catch { - return defaultConfig(); + const parsed: unknown = JSON.parse(readFileSync(path, "utf8")); + if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) throw new Error("config must be an object"); + const raw = parsed as Record; + if (raw.roles !== undefined && (!raw.roles || typeof raw.roles !== "object" || Array.isArray(raw.roles))) { + throw new Error("roles must be an object"); + } + return { ...raw, roles: { ...(raw.roles as Record | undefined) } }; + } catch (err) { + if (forWrite && (err as NodeJS.ErrnoException).code !== "ENOENT") { + throw new Error(`cannot update subagent config: ${err instanceof Error ? err.message : String(err)}`); + } + return { roles: {} }; } - const roles = (parsed && typeof parsed === "object" ? (parsed as { roles?: unknown }).roles : null) as - | Record - | null - | undefined; - const out = defaultConfig(); - if (roles && typeof roles === "object") { - for (const role of ROLES) out.roles[role] = reconstructRole(roles[role]); +} + +function projectTrustWarning(cwd: string, env: NodeJS.ProcessEnv): string | undefined { + if (!isTrackedProjectConfig(cwd)) return undefined; + const token = projectConfigTrustToken(cwd); + if (token !== null && env.CODEXCLAW_TRUST_PROJECT_SUBAGENTS === token) return undefined; + return "ignored Git-tracked .codexclaw/subagents.json; review it, then run `cxc subagents trust-token` and export the printed project-bound value"; +} + +/** Resolve whole roles, preserving explicit null as original-session inheritance. */ +export function readSettings(cwd: string, scope: ConfigScope = "project", env: NodeJS.ProcessEnv = process.env): SubagentSettings { + configScope(scope); + const global = readGlobalRaw(env); + const project = scope === "project" ? readRaw(storePath(cwd)) : { roles: {} }; + const trustWarning = scope === "project" ? projectTrustWarning(cwd, env) : undefined; + const out: SubagentSettings = { + ...defaultConfig(), scope, + sources: { explorer: "session", reviewer: "session", executor: "session" }, + overrides: { explorer: false, reviewer: false, executor: false }, + ...(trustWarning ? { trustWarning } : {}), + }; + for (const role of ROLES) { + out.overrides[role] = Object.hasOwn((scope === "project" ? project : global).roles, role); + if (Object.hasOwn(global.roles, role)) { + out.roles[role] = reconstructRole(global.roles[role]); + out.sources[role] = "global"; + } + if (!trustWarning && Object.hasOwn(project.roles, role)) { + out.roles[role] = reconstructRole(project.roles[role]); + out.sources[role] = "project"; + } } return out; } +/** Effective config without UI metadata; reads never change persisted settings. */ +export function readConfig(cwd: string, scope: ConfigScope = "project", env: NodeJS.ProcessEnv = process.env): SubagentsConfig { + return { roles: readSettings(cwd, scope, env).roles }; +} + /** Validate a role patch, returning an error message or null. */ export function validateRolePatch(patch: Partial): string | null { if (patch.mode !== undefined && patch.mode !== "default" && patch.mode !== "model") { @@ -122,42 +188,48 @@ export function validateRolePatch(patch: Partial): string | null { return null; } -/** Atomic write: temp file then rename. Creates .codexclaw/ if needed. */ -export function writeConfig(cwd: string, config: SubagentsConfig): void { - const dir = join(cwd, STATE_DIR); - if (!existsSync(dir)) mkdirSync(dir, { recursive: true, mode: 0o700 }); - const path = storePath(cwd); - const tmp = `${path}.tmp`; +/** Atomic write with an exclusive temporary file; preserve unrelated JSON fields. */ +function writeRaw(path: string, config: unknown): void { + mkdirSync(dirname(path), { recursive: true, mode: 0o700 }); + const tmp = `${path}.${randomUUID()}.tmp`; try { - writeFileSync(tmp, `${JSON.stringify(config, null, 2)}\n`, { mode: 0o600 }); + writeFileSync(tmp, `${JSON.stringify(config, null, 2)}\n`, { mode: 0o600, flag: "wx" }); renameWithRetry(tmp, path); - } catch (err) { - try { - if (existsSync(tmp)) rmSync(tmp); - } catch { - // best-effort cleanup - } - throw err; + } finally { + rmSync(tmp, { force: true }); } } -/** - * Apply a patch to one role and persist. Returns the updated config. - * Validates the MERGED role (not the bare patch), so `{mode:"model"}` alone is - * valid when the role already has a saved model — the GUI checkbox depends on - * this. Throws on an invalid merged state (caller surfaces the message). - */ -export function setRole(cwd: string, role: RoleName, patch: Partial): SubagentsConfig { +/** Explicit full-config writes remain available to existing callers. */ +export function writeConfig(cwd: string, config: SubagentsConfig): void { + writeRaw(storePath(cwd), config); +} + +/** Merge only the selected role; missing roles continue to inherit dynamically. */ +export function setRole(cwd: string, role: RoleName, patch: Partial, scope: ConfigScope = "project", env: NodeJS.ProcessEnv = process.env): SubagentsConfig { if (!ROLES.includes(role)) throw new Error(`unknown role "${role}"`); - const config = readConfig(cwd); - const next: RoleConfig = { ...config.roles[role], ...patch }; + const path = scopedPath(cwd, scope, env); + const raw = scope === "global" ? readGlobalRaw(env, true) : readRaw(path, true); + const current = Object.hasOwn(raw.roles, role) ? reconstructRole(raw.roles[role]) : readConfig(cwd, scope, env).roles[role]; + const next: RoleConfig = { ...current, ...patch }; const err = validateRolePatch(next); if (err) throw new Error(err); - // enforce the default-mode invariant: default mode ignores model. if (next.mode === "default") next.model = null; - config.roles[role] = next; - writeConfig(cwd, config); - return config; + raw.roles[role] = { ...(typeof raw.roles[role] === "object" && raw.roles[role] !== null ? raw.roles[role] as Record : {}), ...next }; + writeRaw(path, raw); + return readConfig(cwd, scope, env); +} + +/** Remove a role override. null fields deliberately do not perform this action. */ +export function resetRole(cwd: string, role: RoleName, scope: ConfigScope = "project", env: NodeJS.ProcessEnv = process.env): SubagentsConfig { + if (!ROLES.includes(role)) throw new Error(`unknown role "${role}"`); + const path = scopedPath(cwd, scope, env); + const raw = scope === "global" ? readGlobalRaw(env, true) : readRaw(path, true); + if (Object.hasOwn(raw.roles, role)) { + delete raw.roles[role]; + writeRaw(path, raw); + } + return readConfig(cwd, scope, env); } export interface SpawnResolution { @@ -218,20 +290,8 @@ export function resolveSpawnConfig( role: RoleName, env: NodeJS.ProcessEnv = process.env, ): SpawnResolution { - const tracked = isTrackedProjectConfig(cwd); - const expectedTrust = tracked ? projectConfigTrustToken(cwd) : null; - if (tracked && (expectedTrust === null || env.CODEXCLAW_TRUST_PROJECT_SUBAGENTS !== expectedTrust)) { - return { - role, - model: null, - usesMainModel: true, - effort: null, - promptOverride: null, - trustWarning: - "ignored Git-tracked .codexclaw/subagents.json; review it, then run `cxc subagents trust-token` and export the printed project-bound value", - }; - } - const cfg = readConfig(cwd).roles[role]; + const settings = readSettings(cwd, "project", env); + const cfg = settings.roles[role]; const usesMainModel = cfg.mode === "default"; return { role, @@ -239,5 +299,6 @@ export function resolveSpawnConfig( usesMainModel, effort: cfg.effort, promptOverride: cfg.promptOverride, + ...(settings.trustWarning ? { trustWarning: settings.trustWarning } : {}), }; } diff --git a/plugins/codexclaw/components/subagent-config/src/win-exec.ts b/plugins/codexclaw/components/subagent-config/src/win-exec.ts new file mode 100644 index 00000000..5758cd06 --- /dev/null +++ b/plugins/codexclaw/components/subagent-config/src/win-exec.ts @@ -0,0 +1,89 @@ +/** + * win-exec.ts - one entry point for spawning external commands. + * + * Three Windows facts drive this: + * 1. npm-installed CLIs are `.cmd` shims, and Node refuses shell-less `.cmd` spawns + * after the CVE-2024-27980 hardening. + * 2. A bare command name skips PATHEXT resolution, so `spawn("npm")` ENOENTs even + * with `npm.cmd` on PATH (measured, 002 B4). + * 3. `shell: true` is not a fix: Node does not escape cmd metacharacters there, so + * a path containing `&` or `^` becomes a command injection. + * + * Env vars are read case-insensitively: a spawned child can arrive with `Path`, + * `PATH`, or both, and reading one fixed spelling resolves against the wrong list + * (001 3.2). + */ +import { existsSync } from "node:fs"; +import { isAbsolute, join } from "node:path"; + +export interface Invocation { + file: string; + args: string[]; + options: { windowsVerbatimArguments?: boolean }; +} + +/** Case-insensitive env lookup with a key-scan fallback. */ +export function envValue(env: NodeJS.ProcessEnv, name: string): string | undefined { + const direct = env[name]; + if (direct !== undefined) return direct; + const lower = name.toLowerCase(); + for (const key of Object.keys(env)) { + if (key.toLowerCase() === lower) return env[key]; + } + return undefined; +} + +const DEFAULT_PATHEXT = ".COM;.EXE;.BAT;.CMD"; + +/** Resolve `command` against PATH + PATHEXT. Returns the input when nothing matches. */ +export function resolveWindowsCommand(command: string, env: NodeJS.ProcessEnv): string { + if (command.includes("/") || command.includes("\\") || isAbsolute(command)) return command; + const exts = (envValue(env, "PATHEXT") ?? DEFAULT_PATHEXT).split(";").filter((e) => e.length > 0); + // A WINDOWS PATH is always ";"-separated. node:path's `delimiter` follows the + // HOST, so on a Linux runner exercising this win32-only walk it would be ":" + // and the whole PATH would collapse into one bogus directory entry. + const dirs = (envValue(env, "PATH") ?? "").split(";").filter((d) => d.length > 0); + for (const dir of dirs) { + for (const ext of exts) { + const candidate = join(dir, command + ext); + if (existsSync(candidate)) return candidate; + // Case-sensitive filesystems (WSL, Linux CI): PATHEXT spells ".EXE" but + // real shims are "npm.cmd" / "gh.exe". Retry the lowercased extension. + const lowered = join(dir, command + ext.toLowerCase()); + if (existsSync(lowered)) return lowered; + } + } + return command; +} + +/** cross-spawn's escaping: double backslashes before quotes, quote, then caret. */ +function escapeCmdArg(arg: string): string { + let out = arg.replace(/(\\*)"/g, '$1$1\\"').replace(/(\\*)$/, "$1$1"); + out = `"${out}"`; + return out.replace(/[()%!^"<>&|;, ]/g, "^$&"); +} + +function escapeCmdCommand(command: string): string { + return command.replace(/[()%!^"<>&|;, ]/g, "^$&"); +} + +/** + * Build the spawn shape for `command`. POSIX is a passthrough; win32 resolves + * PATHEXT and routes only `.cmd`/`.bat` through cmd.exe. + */ +export function commandInvocation( + command: string, + args: string[], + platform: NodeJS.Platform = process.platform, + env: NodeJS.ProcessEnv = process.env, +): Invocation { + if (platform !== "win32") return { file: command, args: [...args], options: {} }; + const resolved = resolveWindowsCommand(command, env); + if (!/\.(cmd|bat)$/i.test(resolved)) return { file: resolved, args: [...args], options: {} }; + const line = [escapeCmdCommand(resolved), ...args.map(escapeCmdArg)].join(" "); + return { + file: envValue(env, "ComSpec") ?? "cmd.exe", + args: ["/d", "/s", "/c", `"${line}"`], + options: { windowsVerbatimArguments: true }, + }; +} diff --git a/plugins/codexclaw/components/subagent-config/test/catalog.test.ts b/plugins/codexclaw/components/subagent-config/test/catalog.test.ts index 558b2049..77c48b77 100644 --- a/plugins/codexclaw/components/subagent-config/test/catalog.test.ts +++ b/plugins/codexclaw/components/subagent-config/test/catalog.test.ts @@ -69,9 +69,10 @@ test("ocx error mode -> native-catalog (not ocx-active)", () => { assert.equal(cat.state, "native-catalog"); }); -test("native cache absent -> documented fallback set", () => { +test("native cache absent -> unavailable, no fabricated models", () => { const cat = buildCatalog({ readNativeCache: () => null }); - assert.deepEqual(cat.entries.map((e) => e.id), [...NATIVE_OPENAI_MODELS]); + assert.deepEqual(cat.entries, []); + assert.equal(cat.state, "unavailable"); }); test("readNativeCacheDefault: allowlists cache ids, ignores unknowns; missing path -> null", () => { @@ -83,38 +84,36 @@ test("readNativeCacheDefault: allowlists cache ids, ignores unknowns; missing pa const p = join(dir, "models.json"); writeFileSync(p, JSON.stringify({ models: [{ id: "gpt-5.5" }, { id: "rogue-model" }, "gpt-5.4"] })); const ids = readNativeCacheDefault({ CODEX_MODELS_CACHE_PATH: p } as NodeJS.ProcessEnv); - assert.deepEqual(ids, ["gpt-5.5", "gpt-5.4"]); // rogue-model filtered by allowlist + assert.deepEqual(ids, ["gpt-5.5", "rogue-model", "gpt-5.4"]); }); -test("L9.2/L20: readNativeCacheDefault reads slugs; natives by allowlist, routed ocx slugs admitted", () => { +test("L9.2/L20: readNativeCacheDefault reads all configured native and routed slugs", () => { const dir = mkdtempSync(join(tmpdir(), "cxc-cat-slug-")); const p = join(dir, "models.json"); // Live Codex catalog keys natives by slug, not id (opencodex codex-catalog.ts:152,183). - // L20/WP4: opencodex syncs its routed `provider/model` slugs INTO this codex config - // cache, and codexclaw reads them from here (it never calls ocx directly). So routed - // slugs (containing "/") are admitted; a non-native BARE id is still filtered. + // Configured native and routed IDs are authoritative; no compiled allowlist. writeFileSync( p, JSON.stringify({ models: [ { slug: "gpt-5.5", base_instructions: "x" }, { slug: "gpt-5.4-mini" }, - { slug: "gpt-5.3-codex" }, // legacy/internal native bare id -> filtered by allowlist + { slug: "gpt-5.3-codex" }, // configured bare ID retained { slug: "openrouter/grok-4" }, // routed ocx-synced slug -> admitted (L20) { slug: "kiro/claude-opus-4.6" }, // routed ocx-synced slug -> admitted (L20) ], }), ); const ids = readNativeCacheDefault({ CODEX_MODELS_CACHE_PATH: p } as NodeJS.ProcessEnv); - assert.deepEqual(ids, ["gpt-5.5", "gpt-5.4-mini", "openrouter/grok-4", "kiro/claude-opus-4.6"]); + assert.deepEqual(ids, ["gpt-5.5", "gpt-5.4-mini", "gpt-5.3-codex", "openrouter/grok-4", "kiro/claude-opus-4.6"]); }); -test("L20/WP4: a non-native BARE id is still filtered even with routed slugs present", () => { +test("L20/WP4: arbitrary bare IDs are retained alongside routed slugs", () => { const dir = mkdtempSync(join(tmpdir(), "cxc-cat-mix-")); const p = join(dir, "models.json"); writeFileSync(p, JSON.stringify({ models: [{ id: "gpt-5.5" }, { id: "rogue-model" }, { slug: "kiro/claude" }] })); const ids = readNativeCacheDefault({ CODEX_MODELS_CACHE_PATH: p } as NodeJS.ProcessEnv); - assert.deepEqual(ids, ["gpt-5.5", "kiro/claude"]); // rogue-model (bare, non-native) dropped; routed slug kept + assert.deepEqual(ids, ["gpt-5.5", "rogue-model", "kiro/claude"]); }); test("L20/WP4: buildCatalog labels cache-sourced routed slugs as ocx, native bare ids as native", () => { @@ -153,7 +152,7 @@ test("WP30: CODEX_HOME resolution — cache at $CODEX_HOME/models_cache.json loa JSON.stringify({ models: [{ slug: "gpt-5.5" }, { slug: "anthropic/claude-sonnet-5" }, { slug: "rogue" }] }), ); const ids = readNativeCacheDefault({ CODEX_HOME: home } as NodeJS.ProcessEnv); - assert.deepEqual(ids, ["gpt-5.5", "anthropic/claude-sonnet-5"]); // rogue filtered, routed slug admitted + assert.deepEqual(ids, ["gpt-5.5", "anthropic/claude-sonnet-5", "rogue"]); }); test("WP30: explicit CODEX_MODELS_CACHE_PATH still wins over CODEX_HOME", () => { diff --git a/plugins/codexclaw/components/subagent-config/test/global-home.test.ts b/plugins/codexclaw/components/subagent-config/test/global-home.test.ts new file mode 100644 index 00000000..f3add3cf --- /dev/null +++ b/plugins/codexclaw/components/subagent-config/test/global-home.test.ts @@ -0,0 +1,46 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { mkdtempSync, mkdirSync, writeFileSync, readFileSync, rmSync } from 'node:fs'; +import { join } from 'node:path'; +import { tmpdir } from 'node:os'; +import { execFileSync } from 'node:child_process'; +import { cxcHome, globalStorePath, setRole, readSettings } from '../src/store.ts'; + +test('explicit CXC home is independent of Codex home and never imports legacy preferences', t => { + const root = mkdtempSync(join(tmpdir(), 'cxc-home-'));t.after(()=>rmSync(root,{recursive:true,force:true})); + const env={CODEX_HOME:join(root,'codex'),CODEXCLAW_HOME:join(root,'cxc')}; + mkdirSync(join(env.CODEX_HOME,'codexclaw'),{recursive:true}); + writeFileSync(join(env.CODEX_HOME,'codexclaw/subagents.json'),JSON.stringify({roles:{explorer:{mode:'model',model:'legacy'}}})); + assert.equal(cxcHome(env),env.CODEXCLAW_HOME); + assert.equal(globalStorePath(env),join(env.CODEXCLAW_HOME,'subagents.json')); + assert.equal(readSettings(root,'global',env).roles.explorer.model,null); + setRole(root,'reviewer',{effort:'high'},'global',env); + assert.equal(JSON.parse(readFileSync(globalStorePath(env),'utf8')).roles.reviewer.effort,'high'); +}); + +test('legacy global set/reset writes canonical data, preserves other roles, and never resurrects reset roles', t => { + const home=mkdtempSync(join(tmpdir(),'cxc-legacy-'));t.after(()=>rmSync(home,{recursive:true,force:true})); + const env={...process.env,HOME:home,USERPROFILE:home,CODEX_HOME:join(home,'codex')}; + delete env.CODEXCLAW_HOME; + const script=` + import assert from 'node:assert/strict'; + import {mkdirSync,writeFileSync,readFileSync,existsSync} from 'node:fs'; + import {join} from 'node:path'; + import {readSettings,setRole,resetRole,globalStorePath} from ${JSON.stringify(new URL('../src/store.ts',import.meta.url).href)}; + const legacy=join(process.env.CODEX_HOME,'codexclaw/subagents.json'); + mkdirSync(join(process.env.CODEX_HOME,'codexclaw'),{recursive:true}); + const bytes=JSON.stringify({extra:'keep',roles:{explorer:{mode:'model',model:'legacy',effort:null},reviewer:{mode:'model',model:'reviewer',effort:'high'}}}); + writeFileSync(legacy,bytes); + assert.equal(readSettings(process.cwd(),'global').roles.explorer.model,'legacy'); + assert.equal(existsSync(globalStorePath()),false); + resetRole(process.cwd(),'explorer','global'); + assert.equal(readSettings(process.cwd(),'global').roles.explorer.model,null); + assert.equal(readSettings(process.cwd(),'global').roles.reviewer.model,'reviewer'); + setRole(process.cwd(),'reviewer',{effort:null},'global'); + resetRole(process.cwd(),'reviewer','global'); + assert.deepEqual(JSON.parse(readFileSync(globalStorePath(),'utf8')),{extra:'keep',roles:{}}); + assert.equal(readFileSync(legacy,'utf8'),bytes); + assert.equal(readSettings(process.cwd(),'global').roles.explorer.model,null); + `; + execFileSync(process.execPath,['--input-type=module','-e',script],{cwd:home,env}); +}); diff --git a/plugins/codexclaw/components/subagent-config/test/live-catalog.test.ts b/plugins/codexclaw/components/subagent-config/test/live-catalog.test.ts new file mode 100644 index 00000000..cc3b0fe3 --- /dev/null +++ b/plugins/codexclaw/components/subagent-config/test/live-catalog.test.ts @@ -0,0 +1,85 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { mkdtempSync, mkdirSync, writeFileSync, readFileSync, rmSync } from 'node:fs'; +import { join } from 'node:path'; +import { tmpdir } from 'node:os'; +import { execFileSync } from 'node:child_process'; +import { readCatalog, parseOcxModels, runOcxModels, CATALOG_TTL_MS } from '../src/live-catalog.ts'; +import { readNativeCatalog } from '../src/catalog.ts'; + +const roster = (id: string) => JSON.stringify([ + {namespaced:id,native:true,reasoningEfforts:['low','high']}, + {namespaced:'provider/disabled',disabled:true}, + {namespaced:'provider/pending',initialSelectionPending:true}, + {namespaced:'provider/no-effort',reasoningEfforts:[]}, + {namespaced:'provider/unknown',reasoningEfforts:null}, +]); +function fixture(t: {after: (cb:()=>void)=>void}) { + const root=mkdtempSync(join(tmpdir(),'cxc-live-catalog-')); + t.after(()=>rmSync(root,{recursive:true,force:true})); + const env={...process.env,CODEX_HOME:join(root,'codex'),CODEXCLAW_HOME:join(root,'cxc')}; + mkdirSync(env.CODEX_HOME); + return {root,env}; +} + +test('OCX array parsing filters disabled/pending and preserves supported, empty and unknown effort ladders',()=>{ + const rows=parseOcxModels(roster('new-native')); + assert.deepEqual(rows.map(x=>x.id),['new-native','provider/no-effort','provider/unknown']); + assert.deepEqual(rows.map(x=>x.reasoningEfforts),[['low','high'],[],null]); + assert.throws(()=>parseOcxModels('{}'),/invalid OCX/); + assert.throws(()=>parseOcxModels('[{}]'),/invalid OCX/); + assert.deepEqual(parseOcxModels('[]'),[]); +}); + +test('shared last-success cache survives a new process, refresh updates it, failures are labeled stale, empty is authoritative',async t=>{ + const {env}=fixture(t);let now=Date.now(),calls=0,current='first'; + const options={env,now:()=>now,runOcx:async()=>{calls++;return roster(current);}}; + assert.equal((await readCatalog(options)).entries[0].id,'first'); + assert.equal((await readCatalog(options)).entries[0].id,'first');assert.equal(calls,1); + const script=`import {readCatalog} from ${JSON.stringify(new URL('../src/live-catalog.ts',import.meta.url).href)};const c=await readCatalog({runOcx:async()=>{throw Error('cache was not reused')}});console.log(JSON.stringify(c));`; + const child=JSON.parse(execFileSync(process.execPath,['--input-type=module','-e',script],{env,encoding:'utf8'})); + assert.equal(child.status,'fresh');assert.equal(child.entries[0].id,'first'); + current='second';assert.equal((await readCatalog({...options,forceRefresh:true})).entries[0].id,'second'); + now+=CATALOG_TTL_MS+1; + const failed=await readCatalog({...options,runOcx:async()=>{throw new Error('secret stderr never exposed');}}); + assert.equal(failed.status,'stale');assert.equal(failed.entries[0].id,'second');assert.ok(!failed.message?.includes('secret')); + const empty=await readCatalog({...options,forceRefresh:true,runOcx:async()=> '[]'}); + assert.equal(empty.status,'fresh');assert.deepEqual(empty.entries,[]); + assert.deepEqual((await readCatalog(options)).entries,[]); +}); + +test('coalesces concurrent refreshes; malformed payload without cache never invents defaults',async t=>{ + const {env}=fixture(t);let calls=0;let release!:(value:string)=>void; + const runOcx=()=>{calls++;return new Promise(resolve=>release=resolve);}; + const a=readCatalog({env,runOcx,forceRefresh:true}),b=readCatalog({env,runOcx,forceRefresh:true}); + release('[]');await Promise.all([a,b]);assert.equal(calls,1); + rmSync(join(env.CODEXCLAW_HOME,'model-catalog.json')); + const bad=await readCatalog({env,runOcx:async()=> 'not JSON'}); + assert.equal(bad.status,'unavailable');assert.deepEqual(bad.entries,[]); +}); + +test('native fallback honors configured model_catalog_json and arbitrary IDs; explicit path wins; no OCX failure fallback',async t=>{ + const {env}=fixture(t); + writeFileSync(join(env.CODEX_HOME,'config.toml'),`model_catalog_json = 'custom.json' # selected catalog\n[profile]\nmodel_catalog_json = 'wrong.json'\n`); + writeFileSync(join(env.CODEX_HOME,'custom.json'),JSON.stringify({models:[{slug:'gpt-future',supported_reasoning_levels:[{effort:'high'}]},{slug:'hidden',visibility:'hide'}]})); + writeFileSync(join(env.CODEX_HOME,'models_cache.json'),JSON.stringify({models:['stale-default']})); + assert.deepEqual(readNativeCatalog(env)?.map(e=>e.id),['gpt-future']); + const missing=Object.assign(new Error('not installed'),{code:'ENOENT'}); + const native=await readCatalog({env,runOcx:async()=>{throw missing;}}); + assert.equal(native.source,'native');assert.deepEqual(native.entries[0].reasoningEfforts,['high']); + const explicit={...env,CODEX_MODELS_CACHE_PATH:join(env.CODEX_HOME,'models_cache.json')}; + assert.equal(readNativeCatalog(explicit)?.[0].id,'stale-default'); + rmSync(join(env.CODEXCLAW_HOME,'model-catalog.json')); + const failed=await readCatalog({env,runOcx:async()=>{throw new Error('OCX unavailable');}}); + assert.equal(failed.source,'ocx');assert.equal(failed.status,'unavailable'); +}); + +test('real subprocess timeout and oversized stdout are bounded', {skip:process.platform==='win32'}, async t=>{ + const {env,root}=fixture(t);const bin=join(root,'bin');mkdirSync(bin); + const path=join(bin,'ocx');const childEnv={...env,PATH:bin+':'+process.env.PATH}; + writeFileSync(path,'#!/usr/bin/env node\nprocess.stdout.write("x".repeat(5*1024*1024));\n',{mode:0o700}); + await assert.rejects(runOcxModels(childEnv)); + writeFileSync(path,'#!/usr/bin/env node\nsetInterval(()=>{},1000);\n'); + const started=Date.now();await assert.rejects(runOcxModels(childEnv)); + assert.ok(Date.now()-started<18_000,'timeout failed to stop owned subprocess'); +}); diff --git a/plugins/codexclaw/components/subagent-config/test/scoped-surfaces.test.ts b/plugins/codexclaw/components/subagent-config/test/scoped-surfaces.test.ts new file mode 100644 index 00000000..bccb1ce4 --- /dev/null +++ b/plugins/codexclaw/components/subagent-config/test/scoped-surfaces.test.ts @@ -0,0 +1,74 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { mkdtempSync, rmSync } from 'node:fs'; +import { join } from 'node:path'; +import { tmpdir } from 'node:os'; +import { spawnSync } from 'node:child_process'; +import { fileURLToPath } from 'node:url'; +import { setRole } from '../src/store.ts'; + +function fixture(t: { after: (fn: () => void) => void }) { + const cwd = mkdtempSync(join(tmpdir(), 'cxc-surfaces-')); + t.after(() => rmSync(cwd, { recursive: true, force: true })); + const env = { ...process.env, CODEX_HOME: join(cwd, 'codex'), CODEXCLAW_HOME: join(cwd, 'cxc') }; + return { cwd, env }; +} +const script = (name: string) => fileURLToPath(new URL(`../dist/${name}.js`, import.meta.url)); + +test('compiled hook injects global model and effort for v1 and v2; full forks stay untouched', t => { + const { cwd, env } = fixture(t); + setRole(cwd, 'explorer', { mode: 'model', model: 'fixture-global', effort: 'high' }, 'global', env); + for (const tool_input of [{ agent_type: 'explorer', message: 'Explore files' }, { task_name: 'explore', fork_turns: 'none', message: 'Explore files' }]) { + const result = spawnSync(process.execPath, [script('spawn-attach-hook'), 'hook', 'pre-tool-use'], { cwd, env, encoding: 'utf8', input: JSON.stringify({ hook_event_name: 'PreToolUse', tool_name: 'spawn_agent', session_id: 'test-root', cwd, tool_input }) }); + assert.equal(result.status, 0, result.stderr); + const input = JSON.parse(result.stdout).hookSpecificOutput.updatedInput; + assert.equal(input.model, 'fixture-global'); + assert.equal(input.reasoning_effort, 'high'); + } + const result = spawnSync(process.execPath, [script('spawn-attach-hook'), 'hook', 'pre-tool-use'], { cwd, env, encoding: 'utf8', input: JSON.stringify({ hook_event_name: 'PreToolUse', tool_name: 'spawn_agent', session_id: 'test-root', cwd, tool_input: { agent_type: 'explorer', fork_context: true, message: 'Explore files' } }) }); + assert.equal(result.status, 0, result.stderr); + const input = JSON.parse(result.stdout).hookSpecificOutput.updatedInput; + assert.equal(input.model, undefined); + assert.equal(input.reasoning_effort, undefined); +}); + +test('CLI global set/get/reset and project override use the same persisted roles', t => { + const { cwd, env } = fixture(t); + const cli = (...args: string[]) => { + const r = spawnSync(process.execPath, [script('cli'), 'subagents', ...args], { cwd, env, encoding: 'utf8' }); + assert.equal(r.status, 0, r.stdout + r.stderr); return JSON.parse(r.stdout); + }; + assert.equal(cli('set', 'reviewer', '--effort', 'high', '--global').effort, 'high'); + assert.equal(cli('get', 'reviewer').effort, 'high'); + assert.equal(cli('set', 'reviewer', '--clear-effort').effort, null); + assert.equal(cli('get', 'reviewer', '--global').effort, 'high'); + assert.equal(cli('reset', 'reviewer').effort, 'high'); + assert.equal(cli('reset', 'reviewer', '--global').effort, null); +}); + +test('compiled MCP honors scope and reset; rejects invalid effort/scope/reset combinations', t => { + const { cwd, env } = fixture(t); + const args = [ + ['subagents_set', { role: 'reviewer', scope: 'global', effort: 'high' }], + ['subagents_get', {}], + ['subagents_set', { role: 'reviewer', effort: null }], + ['subagents_get', { scope: 'global' }], + ['subagents_set', { role: 'reviewer', inherit: true }], + ['subagents_set', { role: 'reviewer', scope: 'global', effort: 'invalid' }], + ['subagents_set', { role: 'reviewer', scope: 'bad', effort: 'low' }], + ['subagents_set', { role: 'reviewer', inherit: true, effort: null }], + ['subagents_set', { role: 'reviewer', scope: 'global', inherit: true }], + ]; + const input = args.map(([name, arguments_], id) => JSON.stringify({ jsonrpc: '2.0', id, method: 'tools/call', params: { name, arguments: arguments_ } })).join('\n') + '\n'; + const r = spawnSync(process.execPath, [script('mcp')], { cwd, env, input, encoding: 'utf8' }); + assert.equal(r.status, 0, r.stderr); + const replies = r.stdout.trim().split('\n').map(line => JSON.parse(line)); + assert.equal(replies.length, args.length); + const config = (n: number) => JSON.parse(replies[n].result.content[0].text); + assert.equal(config(1).sources.reviewer, 'global'); + assert.equal(config(2).roles.reviewer.effort, null); + assert.equal(config(3).roles.reviewer.effort, 'high'); + assert.equal(config(4).roles.reviewer.effort, 'high'); + for (const n of [5, 6, 7]) assert.equal(replies[n].result.isError, true); + assert.equal(config(8).sources.reviewer, 'session'); +}); diff --git a/plugins/codexclaw/components/subagent-config/test/scopes.test.ts b/plugins/codexclaw/components/subagent-config/test/scopes.test.ts new file mode 100644 index 00000000..63bd9552 --- /dev/null +++ b/plugins/codexclaw/components/subagent-config/test/scopes.test.ts @@ -0,0 +1,103 @@ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { mkdtempSync, mkdirSync, readFileSync, writeFileSync, rmSync } from 'node:fs'; +import { join } from 'node:path'; +import { tmpdir } from 'node:os'; +import { execFileSync } from 'node:child_process'; +import { readSettings, setRole, resetRole, resolveSpawnConfig, projectConfigTrustToken, globalStorePath } from '../src/store.ts'; + +function fixture(t: { after: (fn: () => void) => void }) { + const root = mkdtempSync(join(tmpdir(), 'cxc-scope-')); + t.after(() => rmSync(root, { recursive: true, force: true })); + const cwd = join(root, 'project'); + mkdirSync(cwd); + return { cwd, env: { ...process.env, CODEX_HOME: join(root, 'codex'), CODEXCLAW_HOME: join(root, 'cxc') }, root }; +} + +test('role precedence, explicit null, reset, sparse writes, and project independence', t => { + const { cwd, env, root } = fixture(t); + assert.deepEqual(Object.values(readSettings(cwd, 'project', env).sources), ['session', 'session', 'session']); + setRole(cwd, 'explorer', { mode: 'model', model: 'global-model', effort: 'high' }, 'global', env); + assert.equal(readSettings(cwd, 'project', env).sources.explorer, 'global'); + assert.equal(resolveSpawnConfig(cwd, 'explorer', env).effort, 'high'); + setRole(cwd, 'reviewer', { effort: 'low' }, 'project', env); + const path = join(cwd, '.codexclaw/subagents.json'); + assert.deepEqual(Object.keys(JSON.parse(readFileSync(path, 'utf8')).roles), ['reviewer']); + setRole(cwd, 'explorer', { effort: null }, 'project', env); + assert.equal(readSettings(cwd, 'project', env).roles.explorer.model, 'global-model'); + assert.equal(resolveSpawnConfig(cwd, 'explorer', env).effort, null); + setRole(cwd, 'explorer', { model: 'new-global', effort: 'xhigh' }, 'global', env); + assert.equal(readSettings(cwd, 'project', env).roles.explorer.model, 'global-model'); + const other = join(root, 'other'); mkdirSync(other); + assert.equal(readSettings(other, 'project', env).roles.explorer.model, 'new-global'); + resetRole(cwd, 'explorer', 'project', env); + assert.equal(readSettings(cwd, 'project', env).roles.explorer.effort, 'xhigh'); + resetRole(cwd, 'explorer', 'global', env); + assert.equal(readSettings(cwd, 'project', env).sources.explorer, 'session'); + assert.equal(readSettings(cwd, 'project', env).roles.reviewer.effort, 'low'); +}); + +test('existing complete role configs mask global settings without migration', t => { + const { cwd, env } = fixture(t); + setRole(cwd, 'executor', { mode: 'model', model: 'terra', effort: null }, 'project', env); + const path = join(cwd, '.codexclaw/subagents.json'); + const original = readFileSync(path, 'utf8'); + setRole(cwd, 'executor', { effort: 'high' }, 'global', env); + assert.equal(resolveSpawnConfig(cwd, 'executor', env).model, 'terra'); + assert.equal(resolveSpawnConfig(cwd, 'executor', env).effort, null); + assert.equal(readFileSync(path, 'utf8'), original); +}); + +test('untrusted tracked project falls back to global; project trust still binds bytes', t => { + const { cwd, env } = fixture(t); + execFileSync('git', ['init', '-q', cwd]); + setRole(cwd, 'executor', { mode: 'model', model: 'global' }, 'global', env); + setRole(cwd, 'executor', { model: 'project' }, 'project', env); + execFileSync('git', ['-C', cwd, 'add', '-f', '.codexclaw/subagents.json']); + assert.equal(resolveSpawnConfig(cwd, 'executor', env).model, 'global'); + assert.equal(readSettings(cwd, 'project', env).sources.executor, 'global'); + assert.equal(readSettings(cwd, 'project', env).overrides.executor, true); + assert.match(resolveSpawnConfig(cwd, 'executor', env).trustWarning!, /ignored Git-tracked/); + const trusted = { ...env, CODEXCLAW_TRUST_PROJECT_SUBAGENTS: projectConfigTrustToken(cwd)! }; + assert.equal(resolveSpawnConfig(cwd, 'executor', trusted).model, 'project'); + setRole(cwd, 'executor', { model: 'changed' }, 'project', env); + assert.equal(resolveSpawnConfig(cwd, 'executor', trusted).model, 'global'); +}); + +test('scoped writes preserve unrelated data and reject invalid effort/scope without changing bytes', t => { + const { cwd, env } = fixture(t); + setRole(cwd, 'explorer', { effort: 'low' }, 'global', env); + const path = globalStorePath(env); + const raw = JSON.parse(readFileSync(path, 'utf8')); + raw.extra = { keep: true }; raw.roles.future = { custom: 1 }; + writeFileSync(path, JSON.stringify(raw)); + setRole(cwd, 'reviewer', { effort: null }, 'global', env); + assert.deepEqual(JSON.parse(readFileSync(path, 'utf8')).extra, raw.extra); + assert.deepEqual(JSON.parse(readFileSync(path, 'utf8')).roles.future, raw.roles.future); + const before = readFileSync(path, 'utf8'); + assert.throws(() => setRole(cwd, 'explorer', { effort: 'bad' as never }, 'global', env), /invalid effort/); + assert.throws(() => setRole(cwd, 'explorer', {}, 'bad' as never, env), /invalid scope/); + assert.equal(readFileSync(path, 'utf8'), before); +}); + +test('legacy all-default roles remain explicit overrides until reset', t => { + const { cwd, env } = fixture(t); + const role = { mode: 'default', model: null, effort: null, promptOverride: null }; + mkdirSync(join(cwd, '.codexclaw')); + writeFileSync(join(cwd, '.codexclaw/subagents.json'), JSON.stringify({ roles: { explorer: role, reviewer: role, executor: role } })); + setRole(cwd, 'explorer', { effort: 'high' }, 'global', env); + assert.equal(readSettings(cwd, 'project', env).sources.explorer, 'project'); + assert.equal(readSettings(cwd, 'project', env).roles.explorer.effort, null); + resetRole(cwd, 'explorer', 'project', env); + assert.equal(readSettings(cwd, 'project', env).sources.explorer, 'global'); +}); + +test('malformed persisted state is not overwritten during an edit or reset', t => { + const { cwd, env } = fixture(t); + setRole(cwd, 'explorer', { effort: 'low' }, 'global', env); + const path = globalStorePath(env); + writeFileSync(path, '{broken'); + assert.throws(() => setRole(cwd, 'reviewer', { effort: null }, 'global', env), /cannot update/); + assert.throws(() => resetRole(cwd, 'explorer', 'global', env), /cannot update/); + assert.equal(readFileSync(path, 'utf8'), '{broken'); +}); diff --git a/plugins/codexclaw/components/subagent-config/test/store.test.ts b/plugins/codexclaw/components/subagent-config/test/store.test.ts index 3458df7d..45447cd6 100644 --- a/plugins/codexclaw/components/subagent-config/test/store.test.ts +++ b/plugins/codexclaw/components/subagent-config/test/store.test.ts @@ -169,14 +169,14 @@ test("Git-tracked project config is ignored until the operator explicitly trusts execFileSync("git", ["init", "-q"], { cwd }); execFileSync("git", ["add", "-f", ".codexclaw/subagents.json"], { cwd }); - const denied = resolveSpawnConfig(cwd, "executor", {}); + const denied = resolveSpawnConfig(cwd, "executor", { CODEXCLAW_HOME: join(cwd, "test-global") }); assert.equal(denied.usesMainModel, true); assert.equal(denied.promptOverride, null); assert.match(denied.trustWarning ?? "", /Git-tracked/); const token = projectConfigTrustToken(cwd); assert.ok(token); - const trusted = resolveSpawnConfig(cwd, "executor", { CODEXCLAW_TRUST_PROJECT_SUBAGENTS: token! }); + const trusted = resolveSpawnConfig(cwd, "executor", { CODEXCLAW_HOME: join(cwd, "test-global"), CODEXCLAW_TRUST_PROJECT_SUBAGENTS: token! }); assert.equal(trusted.model, "repo-model"); assert.equal(trusted.promptOverride, "repo instructions"); @@ -185,9 +185,9 @@ test("Git-tracked project config is ignored until the operator explicitly trusts execFileSync("git", ["init", "-q"], { cwd: other }); execFileSync("git", ["add", "-f", ".codexclaw/subagents.json"], { cwd: other }); assert.notEqual(projectConfigTrustToken(other), token, "the same bytes in another repository need separate review"); - assert.equal(resolveSpawnConfig(other, "executor", { CODEXCLAW_TRUST_PROJECT_SUBAGENTS: token! }).model, null); + assert.equal(resolveSpawnConfig(other, "executor", { CODEXCLAW_HOME: join(cwd, "test-global"), CODEXCLAW_TRUST_PROJECT_SUBAGENTS: token! }).model, null); writeFileSync(join(cwd, ".codexclaw", "subagents.json"), JSON.stringify({ roles: {} })); assert.notEqual(projectConfigTrustToken(cwd), token, "editing the reviewed config invalidates trust"); - assert.equal(resolveSpawnConfig(cwd, "executor", { CODEXCLAW_TRUST_PROJECT_SUBAGENTS: token! }).model, null); + assert.equal(resolveSpawnConfig(cwd, "executor", { CODEXCLAW_HOME: join(cwd, "test-global"), CODEXCLAW_TRUST_PROJECT_SUBAGENTS: token! }).model, null); }); diff --git a/plugins/codexclaw/gui/src/App.tsx b/plugins/codexclaw/gui/src/App.tsx index 2da6e039..88e8fb90 100644 --- a/plugins/codexclaw/gui/src/App.tsx +++ b/plugins/codexclaw/gui/src/App.tsx @@ -23,6 +23,7 @@ const NAV: NavItem[] = [ { route: "/agents", label: "Agents", icon: "cpu" }, { route: "/sessions", label: "Sessions", icon: "database" }, { route: "/subagents", label: "Subagents", icon: "sliders" }, + { route: "/settings", label: "Global Settings", icon: "sliders" }, ]; export function App() { @@ -79,7 +80,7 @@ export function App() { ) : active.route === "/sessions" ? ( ) : ( - + )} diff --git a/plugins/codexclaw/gui/src/api.ts b/plugins/codexclaw/gui/src/api.ts index 40645ec7..3aa26a21 100644 --- a/plugins/codexclaw/gui/src/api.ts +++ b/plugins/codexclaw/gui/src/api.ts @@ -9,9 +9,7 @@ */ export type RoleMode = "default" | "model"; -// Mirror of store.ts EFFORTS (separate GUI bundle). Scoped to the universally-supported -// set: codex-rs hard-fails a spawn whose effort is not in the resolved model's -// supported_reasoning_levels, and every selectable model supports exactly these four. +// Store wire values; the UI narrows choices using the selected model capabilities. export const EFFORTS = ["low", "medium", "high", "xhigh"] as const; export type EffortName = (typeof EFFORTS)[number]; @@ -23,11 +21,18 @@ export interface RoleConfig { promptOverride: string | null; } +export type SubagentScope = "project" | "global"; +export type SubagentRole = "explorer" | "reviewer" | "executor"; export interface SubagentsConfig { + scope?: SubagentScope; + sources?: Record; + overrides?: Record; + trustWarning?: string; roles: { explorer: RoleConfig; reviewer: RoleConfig; executor: RoleConfig }; } export interface CatalogEntry { + reasoningEfforts?: string[] | null; id: string; source: "native" | "ocx"; label: string; @@ -122,14 +127,15 @@ export interface SetMultiAgentSurfaceResult { * UI can show the real error instead of a false success. */ export async function setSubagentRole( role: "explorer" | "reviewer" | "executor", - patch: Partial & { role?: never }, + patch: Partial & { role?: never; inherit?: boolean }, fallback: SubagentsConfig, + scope?: SubagentScope, ): Promise { try { const res = await fetch(`${API_BASE}/api/subagents`, { method: "POST", headers: { "content-type": "application/json", "x-codexclaw-local": "1" }, - body: JSON.stringify({ role, ...patch }), + body: JSON.stringify({ role, ...patch, ...(scope ? { scope } : {}) }), }); const body = (await res.json().catch(() => null)) as | (SubagentsConfig & { error?: string }) @@ -138,6 +144,7 @@ export async function setSubagentRole( if (!res.ok || !body || !("roles" in body)) { return { ok: false, config: fallback, error: body?.error ?? `save failed (${res.status})` }; } + if (scope && !isScopedConfig(body, scope)) return { ok: false, config: fallback, error: "Invalid scoped settings response. Reload and try again." }; return { ok: true, config: body as SubagentsConfig }; } catch { return { ok: false, config: fallback, error: "backend unreachable" }; @@ -314,17 +321,45 @@ async function postJson(path: string, body: unknown): Promise<{ ok: boolean; } } +function isScopedConfig(body: unknown, scope: SubagentScope): body is SubagentsConfig { + if (!body || typeof body !== "object") return false; + const b = body as SubagentsConfig; + return b.scope === scope && (["explorer", "reviewer", "executor"] as const).every(role => + b.roles?.[role] && ["project", "global", "session"].includes(b.sources?.[role] ?? "") && typeof b.overrides?.[role] === "boolean"); +} + +/** Settings must load honestly; fabricated defaults could overwrite saved overrides. */ +export async function getSubagentSettings(scope: SubagentScope = "project", signal?: AbortSignal): Promise { + const res = await fetch(`${API_BASE}/api/subagents?scope=${scope}`, { headers: { accept: "application/json", ...LOCAL_HEADER }, signal }); + const body = await res.json(); + if (!res.ok) throw new Error(body?.error ?? `Settings load failed (${res.status})`); + if (!isScopedConfig(body, scope)) throw new Error("Invalid scoped settings response. Update the CXC server and reload."); + return body; +} + +export interface ModelCatalog { + state: string; + entries: CatalogEntry[]; + status: "fresh" | "stale" | "unavailable"; + source: "ocx" | "native"; + fetchedAt: string | null; + message?: string; +} +export async function getModelCatalog(refresh = false): Promise { + try { + const res = await fetch(`${API_BASE}/api/catalog${refresh ? "?refresh=1" : ""}`, { headers: { accept: "application/json", ...LOCAL_HEADER } }); + const body = await res.json(); + if (!res.ok || !body || !Array.isArray(body.entries) || !["fresh", "stale", "unavailable"].includes(body.status)) throw new Error("invalid catalog response"); + return body as ModelCatalog; + } catch { + return { state: "unavailable", entries: [], status: "unavailable", source: "native", fetchedAt: null, message: "Model list could not be loaded. Retry or check the CXC server." }; + } +} + export const api = { getSubagents: () => getJson("/api/subagents", defaultConfig()), getMultiAgentSurface: () => getJson("/api/multi-agent", defaultMultiAgentSurface()), - getCatalog: () => - getJson<{ state: string; entries: CatalogEntry[] }>("/api/catalog", { - state: "native-catalog", - entries: [ - { id: "gpt-5.5", source: "native", label: "gpt-5.5 (native)" }, - { id: "gpt-5.4", source: "native", label: "gpt-5.4 (native)" }, - ], - }), + getCatalog: getModelCatalog, getProvider: () => getJson("/api/provider", { mode: "native", port: null }), // bridge diff --git a/plugins/codexclaw/gui/src/components/EffortSelect.tsx b/plugins/codexclaw/gui/src/components/EffortSelect.tsx index 46199dda..c3a16f77 100644 --- a/plugins/codexclaw/gui/src/components/EffortSelect.tsx +++ b/plugins/codexclaw/gui/src/components/EffortSelect.tsx @@ -3,13 +3,14 @@ import { EFFORTS, type EffortName } from "../api.ts"; interface Props { value: EffortName | null; disabled: boolean; + supported?: readonly string[] | null; onChange: (effort: EffortName | null) => void; } /** Reasoning-effort dropdown. "" = inherit the parent session's effort (null). * Values mirror the codex spawn wire enum; an invalid effort would hard-fail * the spawn, so only these are offered. */ -export function EffortSelect({ value, disabled, onChange }: Props) { +export function EffortSelect({ value, disabled, onChange, supported }: Props) { return ( onChange(e.target.value === "" ? null : e.target.value)} + value={inherited ? "global" : value === null ? "main" : `model:${value}`} + onChange={(e) => { const selected = e.target.value; if (selected === "global") onInherit?.(); else onChange(selected === "main" ? null : selected.slice(6)); }} aria-label="model" > - + + {onInherit ? : null} + {value && !entries.some(e => e.id === value) ? : null} {entries.map((e) => ( - ))} diff --git a/plugins/codexclaw/gui/src/pages/Dashboard.tsx b/plugins/codexclaw/gui/src/pages/Dashboard.tsx index 99b29733..e43606e7 100644 --- a/plugins/codexclaw/gui/src/pages/Dashboard.tsx +++ b/plugins/codexclaw/gui/src/pages/Dashboard.tsx @@ -132,10 +132,13 @@ export function DashboardPage() { toast(`Fallback models use ${version.toUpperCase()} for new sessions`, "ok"); }; - const setRoleModel = async (role: SubagentRole, model: string | null) => { - if (!config || savingRole) return; + const setRoleModel = async (role: SubagentRole, model: string | null, inherit = false) => { + if (!config || savingRole || config.trustWarning) return; + if (!inherit && model && config.roles[role].effort && !(catalog.find(entry => entry.id === model)?.reasoningEfforts ?? []).includes(config.roles[role].effort!)) { + toast("Select session effort in Subagents before choosing this model.", "err"); return; + } setSavingRole(role); - const result = await setSubagentRole(role, { mode: model ? "model" : "default", model }, config); + const result = await setSubagentRole(role, inherit ? { inherit: true } : { mode: model ? "model" : "default", model }, config); setSavingRole(null); if (!result.ok) { toast(result.error ?? `${role} model update failed`, "err"); @@ -266,7 +269,7 @@ function DashboardControls({ savingSurface: boolean; savingRole: SubagentRole | null; onVersionChange: (version: MultiAgentVersion) => void; - onRoleModelChange: (role: SubagentRole, model: string | null) => void; + onRoleModelChange: (role: SubagentRole, model: string | null, inherit?: boolean) => void; }) { const version = surface?.version ?? "v1"; const controlsReady = config !== null; @@ -303,6 +306,7 @@ function DashboardControls({ title="Subagent models" desc="Applies to spawns on both V1 and V2 surfaces when the caller does not pick a model; not applied on full-history forks." > + {config?.trustWarning ?

Project overrides are ignored until trusted. Open Subagents to review or reset them.

: null} {!controlsReady ? ( ) : ( @@ -315,8 +319,10 @@ function DashboardControls({
onRoleModelChange(role, null, true)} value={config.roles[role].mode === "model" ? config.roles[role].model : null} - disabled={savingRole !== null} + disabled={savingRole !== null || !!config.trustWarning} entries={catalog} onChange={(model) => onRoleModelChange(role, model)} /> diff --git a/plugins/codexclaw/gui/src/pages/Subagents.tsx b/plugins/codexclaw/gui/src/pages/Subagents.tsx index 58604c45..ced50e25 100644 --- a/plugins/codexclaw/gui/src/pages/Subagents.tsx +++ b/plugins/codexclaw/gui/src/pages/Subagents.tsx @@ -1,101 +1,134 @@ -import { useEffect, useState } from "react"; -import { - api, - defaultConfig, - setSubagentRole, - type SubagentsConfig, - type CatalogEntry, - type ProviderState, -} from "../api.ts"; +import { useEffect, useRef, useState } from "react"; +import { api, getSubagentSettings, setSubagentRole, type SubagentsConfig, type CatalogEntry, type ProviderState, type SubagentScope, type SubagentRole, type RoleConfig, type ModelCatalog } from "../api.ts"; import { ModelSelect } from "../components/ModelSelect.tsx"; import { EffortSelect } from "../components/EffortSelect.tsx"; -import { PromptOverrideEditor } from "../components/PromptOverrideEditor.tsx"; import { Loading } from "../ui/kit.tsx"; import { toast } from "../ui/toast.tsx"; import { HelpDrawer, HelpTopicButton, useHelp } from "../ui/help.tsx"; const ROLES = ["explorer", "reviewer", "executor"] as const; - -const ROLE_DESC: Record<(typeof ROLES)[number], string> = { +const ROLE_DESC = { explorer: "Read-only search and codebase mapping.", reviewer: "Independent verification and audits.", executor: "Implementation and mutation work.", }; +const SOURCE_LABEL = { project: "Project override", global: "Global defaults", session: "Original session" }; -export function SubagentsPage({ provider }: { provider: ProviderState }) { +export function SubagentsPage({ provider, scope = "project" }: { provider: ProviderState; scope?: SubagentScope }) { + const title = scope === "global" ? "Global Settings" : "Subagents"; const [config, setConfig] = useState(null); const [catalog, setCatalog] = useState([]); - const [savingRole, setSavingRole] = useState(null); + const [catalogState, setCatalogState] = useState(null); + const catalogLoading = useRef(false); + const [refreshing, setRefreshing] = useState(false); + const [savingRole, setSavingRole] = useState(null); + const [error, setError] = useState(null); + const [reload, setReload] = useState(0); + const [prompts, setPrompts] = useState>({ explorer: "", reviewer: "", executor: "" }); + const saving = useRef(false); + const generation = useRef(0); const { helpOpen, helpTopic, openHelp, closeHelp } = useHelp("subagents"); + const dirty = (role: SubagentRole) => prompts[role] !== (config?.roles[role].promptOverride ?? ""); + + async function refreshCatalog(force = false) { + if (catalogLoading.current) return; + catalogLoading.current = true; setRefreshing(true); + const next = await api.getCatalog(force); + setCatalogState(next); setCatalog(next.entries); + catalogLoading.current = false; setRefreshing(false); + } useEffect(() => { - void api.getSubagents().then(setConfig); - void api.getCatalog().then((c) => setCatalog(c.entries)); + void refreshCatalog(); + const onFocus = () => { void refreshCatalog(); }; + window.addEventListener("focus", onFocus); + const interval = window.setInterval(() => void refreshCatalog(), 30_000); + return () => { window.removeEventListener("focus", onFocus); window.clearInterval(interval); }; }, []); + useEffect(() => { + const controller = new AbortController(); + const current = ++generation.current; + setConfig(null); setError(null); + void getSubagentSettings(scope, controller.signal).then(next => { + if (controller.signal.aborted || current !== generation.current) return; + setConfig(next); + setPrompts({ explorer: next.roles.explorer.promptOverride ?? "", reviewer: next.roles.reviewer.promptOverride ?? "", executor: next.roles.executor.promptOverride ?? "" }); + }).catch(err => { + if (!controller.signal.aborted) setError(err instanceof Error ? err.message : "Settings could not be loaded."); + }); + return () => { controller.abort(); generation.current++; }; + }, [scope, reload]); - async function save(role: (typeof ROLES)[number], patch: Partial) { - if (!config) return; - setSavingRole(role); - const result = await setSubagentRole(role, patch, config); - setSavingRole(null); - if (!result.ok) { - toast(result.error ?? `${role} save failed`, "err"); - return; // keep the current config — do not overwrite state with the fallback + async function save(role: SubagentRole, patch: Partial & { inherit?: boolean }) { + if (!config || saving.current) return; + if (patch.mode === "model" && patch.model && config.roles[role].effort !== null) { + const supported = catalog.find(entry => entry.id === patch.model)?.reasoningEfforts ?? []; + if (!supported.includes(config.roles[role].effort!)) { + setError("This model does not support the saved effort. Select session effort first, then choose the model. If using global settings, choose Main model first to customize this role."); return; + } } + saving.current = true; setSavingRole(role); setError(null); + const current = generation.current; + const result = await setSubagentRole(role, patch, config, scope); + saving.current = false; + if (current !== generation.current) return; + setSavingRole(null); + if (!result.ok) { setError(result.error ?? `${role} save failed`); return; } setConfig(result.config); - toast(`${role} updated`, "ok"); + if (patch.inherit || patch.promptOverride !== undefined) { + setPrompts(previous => ({ ...previous, [role]: result.config.roles[role].promptOverride ?? "" })); + } + toast(`${role} ${patch.inherit ? "now inherits defaults" : "updated"}`, "ok"); } return ( <> -
- Subagents - -
+
{title}
-
-

Subagents

-
Per-role model and prompt overrides · saved to .codexclaw/subagents.json
-
- {catalog.length} models · {provider.mode} +

{title}

{scope === "global" ? "Default subagent settings shared by all projects." : "Choose the main model, global settings, or a model for this project."}
+ {catalog.length} models · {catalogState?.source ?? provider.mode}
- {!config ? ( - - ) : ( +
+ {catalogState?.status === "fresh" ? `Model list from ${catalogState.source === "ocx" ? "OCX" : "Codex"}` : catalogState?.status === "stale" ? "Last known model list · refresh failed" : "Model list unavailable"} + +
+ {catalogState?.message ?

{catalogState.message}

: null} + {catalogState?.status === "fresh" && catalog.length === 0 ?

No models are enabled. Enable models in OCX, then refresh.

: null} +

{scope === "global" ? "Projects using Global settings follow these values. Main model uses the original session’s model." : "Global settings follows this role’s defaults, including effort and prompt. Main model uses the original session’s model."}

+ {error ?

{error}

{!config ? : null}
: null} + {config?.trustWarning ?

Project settings are present but ignored until trusted. The values below show the active defaults. Trust the project settings before editing.

{config.trustWarning}

: null} + {!config ? (!error ? : null) : (
- {ROLES.map((role) => { + {ROLES.map(role => { const r = config.roles[role]; - // One fork only: "main model (default)" == default mode; a concrete - // model == model mode. No separate enable-checkbox. + const source = config.sources![role]; + const ignored = scope === "project" && !!config.trustWarning; + const inherited = scope === "project" && !config.overrides![role]; + const supported = r.mode === "model" ? (catalog.find(entry => entry.id === r.model)?.reasoningEfforts ?? null) : undefined; + const unsupported = r.effort !== null && supported !== undefined && !(supported ?? []).includes(r.effort); const effectiveModel = r.mode === "model" ? r.model : null; return ( -
+
{role.charAt(0).toUpperCase() + role.slice(1)} {ROLE_DESC[role]} + {SOURCE_LABEL[source]}{inherited ? ` · ${r.model ?? "main model"} · ${r.effort ?? "session effort"}` : ""}
-
+
- save(role, { mode: model ? "model" : "default", model })} - /> - save(role, { effort })} - /> + void save(role, { inherit: true }) : undefined} value={effectiveModel} disabled={savingRole !== null || ignored} entries={catalog} onChange={model => void save(role, { mode: model ? "model" : "default", model })} /> + void save(role, { effort })} />
- save(role, { promptOverride: v })} - /> -
- {savingRole === role ? saving… : null} -
+ {unsupported ?

Saved effort {r.effort} is not advertised by this model. Select session effort or another supported level.

: null} +