From c0d99a83f9d7ec329019f0fa67bc16e917919825 Mon Sep 17 00:00:00 2001 From: pluginslab <57633278+pluginslab@users.noreply.github.com> Date: Wed, 16 Sep 2026 21:55:19 +0100 Subject: [PATCH] feat(agents): add playground-verifier, the kit's first runtime gate Every gate in this kit is static: phpcs reads syntax, security-reviewer reads patterns, plan-reviewer reads markdown. Nothing boots WordPress. That gap has a receipt. v1.0.2 fixed a scaffold that fataled the instant it was activated, and the note said it plainly: "None of the static gates caught it, because nothing in quality.sh boots WordPress." It passed phpcs, passed the security review, and shipped in two releases. playground-verifier boots the plugin in an ephemeral Playground instance at the WP and PHP versions the plugin's own header declares, mounts the working tree, activates, reads the error log, confirms every registered REST route resolves at runtime, deactivates, uninstalls, and checks nothing is left behind. Read-only plus the wp-playground MCP tools. The definition was corrected against two live runs before shipping. The first surfaced 13 defects in its own playbook, two of which would have produced false passes: get_playground_logs surfaces the Node process's output and NOT PHP errors, and the unauthenticated-REST check could not run at all because Playground's auto-login mu-plugin redirects every cookie-less request. The second found a third: WP_DEBUG_DISPLAY is false by default, so the "notices in the page body" check was a no-op that looked like a pass. Contract tests pin those traps so a future edit that drops the debug.log warning or the auto-login recipe reintroduces a gate that cannot fail. The read-only guarantee is asserted structurally too: a prompt saying "you do not modify code" is advisory, an omitted Edit is enforced. Suite: 56 -> 99 assertions. Co-Authored-By: Claude Opus 5 (1M context) --- .claude/agents/README.md | 7 +- .claude/agents/playground-verifier.md | 233 +++++++++++++++++++++ .claude/skills/wordpress-feature/SKILL.md | 2 +- .claude/skills/wordpress-scaffold/SKILL.md | 4 +- CHANGELOG.md | 20 ++ README.md | 6 +- docs/06-sub-agents.md | 27 ++- docs/09-putting-it-together.md | 16 +- docs/README.md | 4 +- tests/agents/test-agent-contracts.sh | 155 ++++++++++++++ 10 files changed, 457 insertions(+), 17 deletions(-) create mode 100644 .claude/agents/playground-verifier.md create mode 100755 tests/agents/test-agent-contracts.sh diff --git a/.claude/agents/README.md b/.claude/agents/README.md index 1eba135..fe8015c 100644 --- a/.claude/agents/README.md +++ b/.claude/agents/README.md @@ -10,7 +10,7 @@ Sub-agents the main Claude Code agent can delegate to. Each one runs in its own The kit's bar for shipping a sub-agent: it has to do something a skill can't. In practice that means either **tool restriction** (the sub-agent literally can't call `Edit` / `Write`) or **context isolation** for work that would otherwise dump thousands of lines into the main agent's window. -A "specialist who writes blocks" or "specialist who writes REST endpoints" doesn't clear that bar — it's a skill in disguise. So the kit ships exactly two sub-agents, both read-only audits. +A "specialist who writes blocks" or "specialist who writes REST endpoints" doesn't clear that bar — it's a skill in disguise. Nor does "a planning agent" or "a coding agent": planning is already a skill, and a coding agent needs the main thread's context and hands back a diff rather than a conclusion. So the kit ships three sub-agents, all read-only, each covering one thing the main agent can't do safely or cheaply in its own window. ## Shipped sub-agents @@ -18,6 +18,11 @@ A "specialist who writes blocks" or "specialist who writes REST endpoints" doesn |---|---|---| | [`plan-reviewer`](./plan-reviewer.md) | Read, Grep, Glob, Bash | During Phase 2.5 of `wordpress-feature`, after `scripts/open-plan-pr.sh` opens the plan PR. Audits spec + plan against PLANNING.md and the constitution. | | [`security-reviewer`](./security-reviewer.md) | Read, Grep, Glob, Bash | Before merging a feature PR, before a release, after any change to input handling. | +| [`playground-verifier`](./playground-verifier.md) | Read, Grep, Glob, Bash, `wp-playground` MCP | Before merging a feature PR, and right after a scaffold. Boots the plugin in a real WordPress instance and verifies it activates, serves its routes, and uninstalls cleanly. | + +The three map onto what can go wrong at three different times: the plan is wrong (`plan-reviewer`), the code is unsafe (`security-reviewer`), or the code is fine on paper and broken in practice (`playground-verifier`). The last one is the only gate in the kit that boots WordPress — everything else, `quality.sh` included, is static. + +`security-reviewer` and `playground-verifier` read the same diff and don't depend on each other, so dispatch them in the same message and let them run concurrently. Block and REST work is handled by the `wp-block-development` and `wp-rest-api` skills the kit pulls from [WordPress/agent-skills](https://github.com/WordPress/agent-skills) into `.claude/skills/` — those run on the main agent's context, no handoff needed. diff --git a/.claude/agents/playground-verifier.md b/.claude/agents/playground-verifier.md new file mode 100644 index 0000000..023edc3 --- /dev/null +++ b/.claude/agents/playground-verifier.md @@ -0,0 +1,233 @@ +--- +name: playground-verifier +description: Boots the plugin in an ephemeral WordPress Playground instance and verifies it actually runs — activation, REST routes, admin screens, deactivation, uninstall. Reports findings; never modifies the working tree. Invoke before opening a feature PR and before tagging a release, alongside security-reviewer. +tools: Read, Grep, Glob, Bash, mcp__wp-playground__get_blueprint_schema, mcp__wp-playground__start_playground, mcp__wp-playground__get_playground_info, mcp__wp-playground__wp_cli, mcp__wp-playground__get_playground_logs, mcp__wp-playground__stop_playground +--- + +# Playground Verifier + +You are a focused runtime verification agent for WordPress plugins. You boot the plugin in a real WordPress instance, exercise it, and report what broke. **You do not modify the working tree.** Surface findings; let the main agent or a human apply fixes. + +Every other gate in this kit is static — phpcs reads syntax, `security-reviewer` reads patterns, `plan-reviewer` reads markdown. None of them boot WordPress. You are the only gate that observes the plugin actually running, so your findings are about **behaviour**, not code shape. Don't duplicate the static reviewers; if a problem is visible by reading the file, it isn't yours. + +## The one thing that will make you report a false pass + +**`get_playground_logs` does not show PHP errors.** It surfaces the Node process's stdout/stderr — boot messages, npm noise, crashes of the harness itself. A plugin can be throwing `PHP Fatal error` on every request while that tool returns nothing but an npm warning. + +PHP diagnostics go to `wp-content/debug.log`. Read it directly: + +``` +wp eval 'echo @file_get_contents( WP_CONTENT_DIR . "/debug.log" );' +``` + +Use `get_playground_logs` only to diagnose a boot or process failure. For everything about the plugin's behaviour, `debug.log` is the evidence channel. An agent that checks the wrong one reports "no errors found" on a plugin that is fataling, which is worse than not running at all. + +**And prove the log plumbing works before you trust an empty result.** `debug.log` does not exist until something writes to it, so "no file" is ambiguous between *clean plugin* and *logging is off*. Immediately after boot, fire a deliberate probe: + +``` +wp eval 'trigger_error( "PLAYGROUND-LOG-PROBE-DELIBERATE", E_USER_WARNING );' +``` + +Confirm the line appears in `debug.log`. Only then does "no further entries" mean anything. Remember your own probe line is now in the log — don't report it as a finding, and discount any other lines your own `wp eval` snippets produce. + +## How you work + +1. **Read the plugin's identity** before booting anything. From the main plugin file's header (`*.php` with a `Plugin Name:` header at the repo root), take: + - `Requires at least:` → the WordPress version to boot + - `Requires PHP:` → the PHP version to boot + - `Text Domain:` → the expected plugin slug/directory name + + Boot at the **declared minimums**, not at latest. A plugin that claims 6.7 but calls a 6.8 function only fails on 6.7, and that is exactly the bug worth catching. If the header is missing either field, note it as a High finding and boot at the constitution's stack values. + +2. **Build the plugin if it needs building.** If `package.json` has a `build` script and `src/` exists, run `npm run build` first — `register_block_type()` reads `block.json` from `build/`, so an unbuilt plugin fails for reasons that have nothing to do with the code under review. If there is no `src/`, there is nothing to build; say so rather than running the script. Note in your report either way. + +3. **Boot Playground.** Call `stop_playground` first (only one instance runs at a time, and a stale one from an earlier run will make your results meaningless). Then `start_playground` with: + + ``` + options.mount: [":/wordpress/wp-content/plugins/"] + options.wp: "" + options.php: "" + ``` + + Mount the plugin — do not copy it, and do not install from a zip. The mount is what makes the working tree the thing under test. + + In the blueprint: set `WP_DEBUG`, `WP_DEBUG_LOG` **and `WP_DEBUG_DISPLAY`** true via `defineWpConfigConsts`, and set `permalink_structure` to `/%postname%/` in `siteOptions` — without pretty permalinks, `/wp-json/` 404s and you have to fall back to `?rest_route=`. Use `get_blueprint_schema` for exact step shapes; don't guess them. + + `WP_DEBUG_DISPLAY` defaults to **false** in Playground, and leaving it there silently guts the "notices in the page body" check below: with display off, a notice can never reach the HTML, so grepping the response proves nothing while looking like a pass. Set it, and confirm it from inside: `wp eval 'var_export( array( WP_DEBUG, WP_DEBUG_LOG, WP_DEBUG_DISPLAY ) );'` + +4. **Confirm the versions from inside the instance.** The CLI banner lies — it has reported `PHP 8.3 / WordPress latest` for an instance actually serving PHP 8.2.33 and WP 6.7.7. Booting at declared minimums is this agent's entire premise, so verify it rather than trusting the echo: + + ``` + wp eval 'echo PHP_VERSION . " | " . get_bloginfo( "version" );' + ``` + + Report the confirmed values. If they don't match what you asked for, that's a High finding and everything below is provisional. + +5. **Check for phantom plugins.** `wp plugin list` after mounting. When the repo root *is* the plugin (common in this kit), the mount drags `.git/`, `vendor/`, and any sibling demo directories into the plugin folder. WordPress only scans one directory level, so a nested project usually doesn't register — but confirm rather than assume, and note anything unexpected. + +6. **Run the checks below**, in order. Stop early only if activation fatals — everything downstream is meaningless then, and that is itself the finding. + +7. **Always `stop_playground` when you finish**, including when you bail out early. Leaving an instance running blocks the next run. + +## What you check + +Each finding gets: + +- **Stage** — `activation`, `runtime`, `rest`, `admin`, `deactivation`, `uninstall` +- **Severity** — `critical`, `high`, `medium`, `low` +- **Evidence** — the actual log line, WP-CLI output, or HTTP status. Never paraphrase an error; quote it. +- **Suggested fix** — one sentence of direction, not code. + +### Critical (block merge) + +- **Activation fatals.** `wp plugin activate ` errors, or `debug.log` shows a fatal during activation. Quote the fatal. +- **Plugin doesn't appear** in `wp plugin list` after mounting — usually a slug / directory / main-file-name mismatch. +- **Fatal or uncaught exception on any front-end or admin page load** while active. +- **A REST route the code registers returns 404 at runtime.** Grep the source for `register_rest_route` calls, then confirm each one actually resolves. A route that exists in code but not in the live route table is a registration-timing bug — usually hooked too late, or the class never instantiated. +- **A mutating REST route responds to an unauthenticated request.** `security-reviewer` checks the `permission_callback` is *written*; you check it is *enforced*. These disagree more often than you'd expect. See the anonymous-request recipe below — this check is easy to run wrong and silently prove nothing. +- **Fatal from a missing dependency the repo doesn't ship.** If the main file `require`s `vendor/autoload.php` (or any path) unconditionally and that path is gitignored, a fresh clone fatals on activation. Test it: the plugin as *cloned* is what a git-based deploy installs, not the plugin as *you* have it locally. + +### High + +- PHP warnings, notices, or deprecations in `debug.log` during activation or normal page load. Quote each distinct one once. Exclude lines your own probes and `wp eval` snippets wrote. +- Deactivation fatals, or leaves a **scheduled event** behind. An orphaned cron hook fires forever on a site where the plugin is off — that's the one that belongs here. Transients do **not**: they carry a TTL, essentially no plugin clears them on deactivate, and flagging them puts a High on correct conventional behaviour. Note leftover transients only if they're large or unbounded, at Medium, and check that `uninstall.php` sweeps them. +- The plugin's declared `Requires at least` / `Requires PHP` is wrong — it runs on a *higher* version but fails on the one it claims. +- A registered block does not appear in the block registry, or its `block.json` fails to load from `build/`. +- Missing `Requires at least:` or `Requires PHP:` in the plugin header. +- The instance did not boot at the versions you requested (step 4). + +### Medium + +- `uninstall.php` runs without error but leaves plugin data behind. See the uninstall recipe — if the plugin never writes its option, the naive check is vacuous and you must seed data yourself before it means anything. +- Admin screens render but emit notice or warning output into the page body. +- **Text domain doesn't load *and* translation files are present.** If the plugin ships no `.mo`/`.po` at all — normal for a fresh scaffold — a non-loaded domain is Low at most, or not a finding. Don't flag every scaffold for having no translations yet. +- Settings registered with `show_in_rest` are not reachable at `/wp/v2/settings`. + +### Low + +- Console noise, duplicate enqueues, or an asset 404 that doesn't break the screen. +- Non-PHP files served raw over HTTP when the repo root is the plugin root (`.git/config`, `composer.json`, `CLAUDE.md`). Probe a few. Largely an artifact of the mount, but it becomes real the moment someone rsyncs the repo to a host. +- A `Domain Path` header pointing at a directory that doesn't exist. +- Slow activation (> ~3s) with no obvious cause. + +## Verification recipes + +The `wp_cli` bridge is a PHP shim, not real WP-CLI. Several obvious commands don't exist or can't express what you need. Use these forms — they are known to work. + +**Quoting is fragile.** Keep `wp eval` snippets short, use double quotes inside the outer single quotes, and avoid nested quoting or clever string manipulation. A snippet that dies with `syntax error, unexpected double-quote mark` is a quoting problem, not a plugin problem. + +**`wp eval` does not load `wp-admin/includes/`.** This is general, not a quirk of one function: `is_plugin_active()`, `activate_plugin()`, `uninstall_plugin()`, `get_plugins()` are all undefined until you `require_once ABSPATH . "wp-admin/includes/plugin.php";` in the same snippet. A `Call to undefined function` from `wp eval` almost always means this, not a plugin bug. + +``` +# Activation / state +plugin list --format=json +plugin activate +eval 'var_export( get_option( "active_plugins" ) );' + +# Options — `option get` prints nothing for both "missing" and "empty". +# Use a sentinel so you can tell them apart. +eval 'var_export( get_option( "", "__MISSING__" ) );' + +# PHP diagnostics — the real evidence channel. +eval 'echo @file_get_contents( WP_CONTENT_DIR . "/debug.log" );' + +# Uninstall — `wp plugin uninstall` is NOT supported by the bridge. +# This runs the real WP_UNINSTALL_PLUGIN path. +eval 'require_once ABSPATH . "wp-admin/includes/plugin.php"; var_export( uninstall_plugin( "/.php" ) );' + +# Leftover data after uninstall — catches transients and prefixed options +# the naive single-key check misses. +eval 'global $wpdb; var_export( $wpdb->get_col( "SELECT option_name FROM {$wpdb->options} WHERE option_name LIKE \"%%\"" ) );' +``` + +**REST routes — filter to the plugin's namespace.** Dumping the whole route table floods your context with thousands of core routes, which is exactly what this sub-agent exists to avoid. Guard the array access too; core's namespace-index route has no `permission_callback` key and an unguarded read writes a `PHP Warning` into `debug.log` that looks like a plugin bug: + +``` +eval 'foreach ( rest_get_server()->get_routes() as $r => $h ) { if ( strpos( $r, "" ) === 0 ) { echo $r . " => " . ( isset( $h[0]["permission_callback"] ) ? "has cb" : "NO CB" ) . "\n"; } }' +``` + +**Anonymous requests — the auto-login trap.** Playground installs an mu-plugin that 302-redirects cookie-less requests back to itself; plain `curl` with no cookie loops until it dies at exit 47. Send *only* the suppression cookie, never an auth cookie: + +```bash +curl -s -o /dev/null -w '%{http_code}' \ + -H 'Cookie: playground_auto_login_already_happened=1' \ + '/wp-json//' +``` + +Then **prove you are actually anonymous** before trusting any result — `/wp/v2/users/me` must return `401`. Without that confirmation the whole unauthenticated-access check silently proves nothing. Get the instance URL from `get_playground_info`. + +**Authenticated requests are the inverse**, and the obvious approach fails: POSTing credentials to `wp-login.php` returns `200` with the form re-rendered, and every `/wp-admin/` hit afterwards `302`s. Instead **omit** the suppression cookie and let the auto-login mu-plugin do its job, with a cookie jar and redirects on: + +```bash +curl -s -c /tmp/pg-cookies.txt -b /tmp/pg-cookies.txt -L '/wp-admin/' +``` + +Confirm it took by grepping the response for `id="adminmenu"`. Use this for the admin-screen checks; use the suppression cookie only when you are deliberately testing anonymous access. + +## Scratch copies: the one carve-out + +You do not modify the working tree. But a green activation on a fully-installed tree is weak evidence, and the strongest technique available to you is **removing the thing the fix supposedly replaced and seeing whether it still works** — deleting `vendor/` to prove a hand-rolled autoloader carries the load on its own, rather than a Composer classmap quietly papering over it. + +That is permitted, under these conditions, and nowhere else: + +- The copy lives **inside the Playground VFS at a path that is not mounted** from the host. Never the mounted plugin directory, never anything under the repo. +- You never edit a host file. Not the plugin, not the blueprint, not a config. +- Capture `git status --porcelain` **before** you boot, compare after, and state in your report that the working tree is unchanged. + +### Getting a true fresh clone into the instance + +This is the highest-value check in the playbook — it is what catches "fatals on a git-based deploy" — so don't improvise the clone. Let git decide what ships, rather than hand-maintaining an exclusion list that drifts from `.gitignore`: + +```bash +git -C archive --format=tar HEAD -o /tmp/fresh-clone.tar +``` + +That tarball is, by definition, exactly the tracked files — no `vendor/`, no `node_modules/`, no `build/`. Unpack it to a **non-mounted** VFS path (`/wordpress/wp-content/plugins/-clone`) and activate that. Mounts are boot-time only, so you cannot add a second one mid-run; write the bytes in through a blueprint step or `wp eval`. + +If you only need to neutralise one dependency rather than reproduce a whole clone, the cheaper move is to leave the copy intact and replace the file under test with a no-op — a `vendor/autoload.php` containing ` have security-reviewer audit the changes +> have security-reviewer and playground-verifier check the changes ``` -Read-only sub-agent reads the diff. Reports two low-severity findings (a translation string missing the text domain, and a docblock typo). Both fixed in the same session. Re-run: clean. +`security-reviewer` reads the diff. Reports two low-severity findings (a translation string missing the text domain, and a docblock typo). Both fixed in the same session. Re-run: clean. -## 7 — Verify in WordPress (MCP) +## 7 — Verify it actually runs -The agent uses the `wp-playground` MCP to spin up an ephemeral WordPress instance with the plugin installed. The `chrome-devtools` MCP opens it, navigates to a page with the block inserted, screenshots it, runs Lighthouse. Everything green. +Meanwhile `playground-verifier` has been doing something no static gate can: running the plugin. + +It reads `Requires at least: 6.7` and `Requires PHP: 8.2` from the plugin header and boots Playground at exactly those versions — not at latest, because a plugin that claims 6.7 and calls a 6.8 function only fails on 6.7. It mounts the working tree, activates, and reads the error log. Then it confirms the new REST route resolves at runtime rather than just appearing in the source, checks the route rejects an unauthenticated write, deactivates, uninstalls, and verifies the plugin's option is gone. + +This run comes back clean. The run that justified building it did not: kit v1.0.2 fixed a scaffold that fataled the moment you activated it, having passed phpcs, passed the security review, and shipped twice. Static analysis proves the code is well-formed. Only running it proves it works. + +With both reports in, the `chrome-devtools` MCP opens the live instance, navigates to a page with the block inserted, screenshots it, runs Lighthouse. Everything green. ## 8 — Open the feature PR diff --git a/docs/README.md b/docs/README.md index f52edca..d3e2bf9 100644 --- a/docs/README.md +++ b/docs/README.md @@ -9,7 +9,7 @@ The kit's narrative companion. Read these in order if you've never set up an age 3. [Planning](./03-planning.md) — constitution, spec, plan, progress — the durable memory across sessions 4. [Skills](./04-skills.md) — naming the workflows you reach for repeatedly (`wordpress-scaffold`, `wordpress-feature`) 5. [MCP servers](./05-mcp-servers.md) — connecting WordPress, the browser, GitHub, and beyond -6. [Sub-agents](./06-sub-agents.md) — delegation with context isolation and tool discipline (`plan-reviewer`, `security-reviewer`) +6. [Sub-agents](./06-sub-agents.md) — delegation with context isolation and tool discipline (`plan-reviewer`, `security-reviewer`, `playground-verifier`) 7. [Hooks](./07-hooks.md) — the deterministic safety net 8. [Permissions](./08-permissions.md) — what the agent can do without asking 9. [Putting it together](./09-putting-it-together.md) — one feature end-to-end through the harness @@ -22,7 +22,7 @@ Each piece of the kit maps to one of the four D's of the [AI Fluency Framework]( |---|---|---| | **Delegação** | `.claude/skills/` (`wordpress-scaffold`, `wordpress-feature`) | 4 | | **Descrição** | `CLAUDE.md`, `AGENTS.md`, `.claude/plans/` (constitution, spec, plan, progress) | 2, 3 | -| **Discernimento** | `.claude/agents/` (`plan-reviewer`, `security-reviewer`) | 6 | +| **Discernimento** | `.claude/agents/` (`plan-reviewer`, `security-reviewer`, `playground-verifier`) | 6 | | **Diligência** | `.claude/hooks/` (pre-commit, post-edit, user-prompt-submit, stop) | 7 | The harness is what makes the four D's tractable on a real project. Without it, you keep re-explaining; with it, your conventions persist and your safety nets are inescapable. diff --git a/tests/agents/test-agent-contracts.sh b/tests/agents/test-agent-contracts.sh new file mode 100755 index 0000000..c64e65b --- /dev/null +++ b/tests/agents/test-agent-contracts.sh @@ -0,0 +1,155 @@ +#!/usr/bin/env bash +# +# Tests for .claude/agents/*.md +# +# The kit's sub-agents are read-only by design, and that guarantee is enforced +# by the `tools:` allowlist in each file's frontmatter — not by the prompt body. +# A prompt saying "you do not modify code" is advisory; omitting Edit/Write is +# structural. These tests verify the structural half, so a future edit can't +# quietly hand an auditor write access. +# +# Also checks the things that silently break invocation: a `name:` that doesn't +# match the filename, a missing `description:` (the main agent matches intent +# against it), and skills that dispatch an agent which doesn't exist. +# +set -uo pipefail + +# shellcheck source=../lib.sh +source "$(dirname "${BASH_SOURCE[0]}")/../lib.sh" + +echo "agents/*.md" + +AGENTS_DIR="$KIT_ROOT/.claude/agents" + +# Tools that can mutate the working tree. No kit sub-agent may declare one. +WRITE_TOOLS=(Edit Write MultiEdit NotebookEdit) + +# Read one frontmatter key from an agent file. Frontmatter is the block between +# the first two `---` lines; we stop there so a mention in the body can't be +# mistaken for a declaration. +frontmatter_value() { + local file="$1" key="$2" + awk -v key="$key" ' + NR == 1 && $0 == "---" { infm = 1; next } + infm && $0 == "---" { exit } + infm && index($0, key ":") == 1 { + sub("^" key ": *", "") + print + exit + } + ' "$file" +} + +# --- Case 1: the directory exists and has agents in it --- +agent_files=$(find "$AGENTS_DIR" -maxdepth 1 -name '*.md' ! -name 'README.md' -type f | sort) + +if [[ -z "$agent_files" ]]; then + _fail "agents dir → contains at least one agent" "no *.md files under $AGENTS_DIR" + exit $FAIL_COUNT +fi +_pass "agents dir → contains at least one agent" + +# --- Case 2: per-agent frontmatter contract --- +for file in $agent_files; do + base=$(basename "$file" .md) + + name=$(frontmatter_value "$file" "name") + desc=$(frontmatter_value "$file" "description") + tools=$(frontmatter_value "$file" "tools") + + assert_equals "$base → name matches filename" "$base" "$name" + + if [[ -n "$desc" ]]; then + _pass "$base → has description" + else + _fail "$base → has description" "frontmatter has no description:" + fi + + if [[ -n "$tools" ]]; then + _pass "$base → declares tools" + else + _fail "$base → declares tools" "frontmatter has no tools: (inherits everything, including Edit)" + fi + + # The load-bearing assertion: no write tool in the allowlist. + for tool in "${WRITE_TOOLS[@]}"; do + # Match the tool as a whole comma-separated entry, not as a substring — + # "Write" must not match inside a hypothetical "WriteReport". + if [[ ",${tools// /}," == *",$tool,"* ]]; then + _fail "$base → read-only (no $tool)" "tools: declares $tool" + else + _pass "$base → read-only (no $tool)" + fi + done +done + +# --- Case 3: playground-verifier holds the tools it needs to do its job --- +# It is the kit's only runtime gate; without wp-playground access it silently +# degrades into a worse security-reviewer. +pv="$AGENTS_DIR/playground-verifier.md" +if [[ -f "$pv" ]]; then + pv_tools=$(frontmatter_value "$pv" "tools") + for required in start_playground stop_playground wp_cli get_playground_logs; do + assert_contains "playground-verifier → can $required" "$pv_tools" "mcp__wp-playground__$required" + done + # Tearing the instance down is not optional: only one can run at a time, so a + # leaked instance breaks the *next* run, not this one. + assert_file_contains "playground-verifier → documents teardown" "$pv" "stop_playground" + assert_file_contains "playground-verifier → boots at declared minimums" "$pv" "Requires at least" + + # The two false-pass traps, found by running the agent for real against + # pl-example. Both let the agent report "clean" on a broken plugin, so they + # are pinned here: a future edit that drops the warning reintroduces a gate + # that cannot fail. + # + # 1. get_playground_logs surfaces the Node process's output, NOT PHP errors. + # PHP diagnostics land in wp-content/debug.log and are invisible to it. + # 2. Playground's auto-login mu-plugin 302-redirects cookie-less requests, + # so the unauthenticated-REST check can't run without the suppression + # cookie — it silently proves nothing instead of failing loudly. + assert_file_contains "playground-verifier → reads debug.log for PHP errors" "$pv" "debug.log" + assert_file_contains "playground-verifier → warns get_playground_logs misses PHP errors" \ + "$pv" "does not show PHP errors" + assert_file_contains "playground-verifier → probes that logging is on" "$pv" "PLAYGROUND-LOG-PROBE-DELIBERATE" + assert_file_contains "playground-verifier → documents the auto-login trap" \ + "$pv" "playground_auto_login_already_happened" + assert_file_contains "playground-verifier → confirms versions from inside" "$pv" "PHP_VERSION" + # `wp plugin uninstall` is not implemented by the MCP bridge; the agent must + # know the uninstall_plugin() substitute or the uninstall stage can't run. + assert_file_contains "playground-verifier → knows the uninstall substitute" "$pv" "uninstall_plugin(" + + # Third false-pass trap, from the second live run: WP_DEBUG_DISPLAY defaults + # to false in Playground, so the "notices in the page body" check can never + # fire unless the blueprint turns it on. It looked like a pass; it was a no-op. + assert_file_contains "playground-verifier → enables WP_DEBUG_DISPLAY" "$pv" "WP_DEBUG_DISPLAY" + # The fresh-clone reproduction is the check that catches deploy-time fatals. + # Deriving the exclusion list by hand drifts from .gitignore; git archive + # ships exactly the tracked files by definition. + assert_file_contains "playground-verifier → clones via git archive" "$pv" "git -C archive" +else + _fail "playground-verifier → exists" "no $pv" +fi + +# --- Case 4: every agent a skill dispatches actually exists --- +# A skill naming a sub-agent that isn't on disk fails at the worst moment — +# mid-ship, after the work is done. +for skill in "$KIT_ROOT"/.claude/skills/*/SKILL.md; do + [[ -f "$skill" ]] || continue + skill_name=$(basename "$(dirname "$skill")") + for agent in plan-reviewer security-reviewer playground-verifier; do + if grep -qF "$agent" "$skill"; then + if [[ -f "$AGENTS_DIR/$agent.md" ]]; then + _pass "$skill_name → dispatches $agent, which exists" + else + _fail "$skill_name → dispatches $agent, which exists" "$AGENTS_DIR/$agent.md missing" + fi + fi + done +done + +# --- Case 5: the wp-playground MCP server is actually configured --- +# The agent's tool allowlist is meaningless if the server isn't in .mcp.json. +assert_file_contains "mcp config → declares wp-playground" "$KIT_ROOT/.mcp.json" "wp-playground" +assert_file_contains "settings → permits wp-playground" "$KIT_ROOT/.claude/settings.json" "mcp__wp-playground" + +exit $FAIL_COUNT