From c4338091195cc60ef3807735f104d2393f1efabb Mon Sep 17 00:00:00 2001 From: pluginslab <57633278+pluginslab@users.noreply.github.com> Date: Sat, 1 Aug 2026 10:25:05 +0100 Subject: [PATCH 1/5] fix: reliability pass on the ReAct loop (structured output, summarize, stale docs) Three independent fixes, all in the inference path. 1. Correct stale docs and delete dead config. ARCHITECTURE.md and .claude/CLAUDE.md claimed a 4096-token context window and 2000-char tool-result truncation. Both are wrong: MODEL_CONTEXT_SIZES sets 8192 for Qwen3 1.7B and 32768 for Qwen2.5 7B, and maxToolResultLength is 3000. MODEL_CONFIG.context_window_size looked authoritative, contradicted MODEL_CONTEXT_SIZES, and was never passed to the engine; its only reference was its own export. Removed. 2. Grammar-constrain the ReAct action envelope. The loop asked for JSON and then repaired whatever came back. It now passes REACT_ACTION_SCHEMA as response_format so the grammar enforces the envelope during decoding. Gated on thinking being disabled, because a JSON grammar cannot represent Qwen 3's leading block. ExternalEngine strips the WebLLM-specific `schema` key so OpenAI-compatible providers still get plain JSON mode. Escape hatch: structuredOutput: false. 3. Adopt preferSummarize on the four report abilities. site-health, security-scan, verify-core-checksums and file-scan all render complete reports in summarize(), then had them re-rendered by the LLM against a 3000-char cap. A truncated security scan is worse than none. Workflows do not consult preferSummarize, so the four workflows chaining these abilities are unaffected. Prerequisite for 3: summarize() emitted broken markdown bold (`* * 8.2 * *` instead of `**8.2**`) in site-health, core-site-info, core-environment-info and plugin-list. The LLM was silently repairing it; rendering summarize() directly would have shipped it to users verbatim. Tests: 4 new cases covering the gating both ways plus the escape hatch. 95 unit tests pass, lint clean, production build succeeds. Refs #228 --- .claude/CLAUDE.md | 4 +- docs/ARCHITECTURE.md | 5 +- .../abilities/core-environment-info.js | 10 +-- src/extensions/abilities/core-site-info.js | 14 +-- src/extensions/abilities/file-scan.js | 5 ++ src/extensions/abilities/plugin-list.js | 4 +- src/extensions/abilities/security-scan.js | 5 ++ src/extensions/abilities/site-health.js | 49 +++++----- .../abilities/verify-core-checksums.js | 5 ++ .../services/__tests__/react-agent.test.js | 89 +++++++++++++++++++ src/extensions/services/external-engine.js | 8 ++ src/extensions/services/model-loader.js | 9 -- src/extensions/services/react-agent.js | 46 ++++++++++ 13 files changed, 201 insertions(+), 52 deletions(-) diff --git a/.claude/CLAUDE.md b/.claude/CLAUDE.md index 25a76e1..cb78d69 100644 --- a/.claude/CLAUDE.md +++ b/.claude/CLAUDE.md @@ -58,8 +58,8 @@ Update ALL of these: ## Constraints -- **Context window**: 4096 tokens (1.7B) / 32768 (7B). Tool descriptions sent every request — keep them concise. -- **Tool results truncated** to 2000 chars before sending to LLM. +- **Context window**: 8192 tokens (1.7B) / 32768 (7B), 8192 default. Source of truth is `MODEL_CONTEXT_SIZES` in `model-loader.js`; override per model via the `agentic_admin_context_size` localStorage key. Tool descriptions sent every request — keep them concise. +- **Tool results truncated** to `maxToolResultLength` (3000 chars) before sending to LLM. Abilities with `preferSummarize: true` bypass this entirely. - **Max 10 ReAct iterations**. Repeated tool call detection stops oscillation. - **Service Worker** (`sw.js`) must be self-contained — no code splitting, no dynamic imports. diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index e40fe3b..838860d 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -319,13 +319,14 @@ The 1.7B model is recommended for most users — it loads faster, uses less VRAM 1. **JSON formatting** - 7B models produce valid JSON consistently (100% parse success with Qwen2.5-7B). Robust parsing still handles edge cases: try native `JSON.parse` first, then fall back to quote sanitization. 2. **Goal efficiency** - Qwen2.5-7B calls exactly 1 tool for single-goal tasks (no over-shooting). 3. **Multi-step reasoning** - 100% success on conditional logic and diagnose-then-fix chains. -4. **Context limits** - 4096 token context window configured (models support up to 32K). +4. **Context limits** - Context window is set per model in `MODEL_CONTEXT_SIZES` (`model-loader.js`): 8192 tokens for Qwen3 1.7B, 32768 for Qwen2.5 7B, 8192 default fallback. A per-model override can be stored in the `agentic_admin_context_size` localStorage key and is resolved by `ModelLoader.getEffectiveContextSize()`. ### Safety Mechanisms - Repeated call detection (same tool twice = stop) - Max 10 iterations -- Tool result truncation (2000 chars max in prompt-based mode) +- Tool result truncation (`maxToolResultLength`, 3000 chars, in prompt-based mode) +- Schema-constrained JSON for the ReAct action envelope when thinking is disabled (see below) - Context window overflow handling - JSON envelope unwrapping (prevents raw `{"action": "final_answer", ...}` leaking to user) diff --git a/src/extensions/abilities/core-environment-info.js b/src/extensions/abilities/core-environment-info.js index 5bdd609..47f9ff6 100644 --- a/src/extensions/abilities/core-environment-info.js +++ b/src/extensions/abilities/core-environment-info.js @@ -71,18 +71,16 @@ export function registerCoreEnvironmentInfo() { const envDisplay = result.environment.charAt( 0 ).toUpperCase() + result.environment.slice( 1 ); - lines.push( ` * * Environment: * * ${ envDisplay }` ); + lines.push( `**Environment:** ${ envDisplay }` ); } if ( result.wp_version ) { - lines.push( - ` * * WordPress Version: * * ${ result.wp_version }` - ); + lines.push( `**WordPress Version:** ${ result.wp_version }` ); } if ( result.php_version ) { - lines.push( ` * * PHP Version: * * ${ result.php_version }` ); + lines.push( `**PHP Version:** ${ result.php_version }` ); } if ( result.db_server_info ) { - lines.push( ` * * Database: * * ${ result.db_server_info }` ); + lines.push( `**Database:** ${ result.db_server_info }` ); } if ( lines.length === 0 ) { diff --git a/src/extensions/abilities/core-site-info.js b/src/extensions/abilities/core-site-info.js index 03801e4..042f323 100644 --- a/src/extensions/abilities/core-site-info.js +++ b/src/extensions/abilities/core-site-info.js @@ -110,25 +110,25 @@ export function registerCoreSiteInfo() { const lines = []; if ( result.name ) { - lines.push( ` * * Site Name: * * ${ result.name }` ); + lines.push( `**Site Name:** ${ result.name }` ); } if ( result.description ) { - lines.push( ` * * Tagline: * * ${ result.description }` ); + lines.push( `**Tagline:** ${ result.description }` ); } if ( result.url ) { - lines.push( ` * * Site URL: * * ${ result.url }` ); + lines.push( `**Site URL:** ${ result.url }` ); } if ( result.version ) { - lines.push( ` * * WordPress Version: * * ${ result.version }` ); + lines.push( `**WordPress Version:** ${ result.version }` ); } if ( result.language ) { - lines.push( ` * * Language: * * ${ result.language }` ); + lines.push( `**Language:** ${ result.language }` ); } if ( result.admin_email ) { - lines.push( ` * * Admin Email: * * ${ result.admin_email }` ); + lines.push( `**Admin Email:** ${ result.admin_email }` ); } if ( result.charset ) { - lines.push( ` * * Charset: * * ${ result.charset }` ); + lines.push( `**Charset:** ${ result.charset }` ); } if ( lines.length === 0 ) { diff --git a/src/extensions/abilities/file-scan.js b/src/extensions/abilities/file-scan.js index c3390b6..3a55e74 100644 --- a/src/extensions/abilities/file-scan.js +++ b/src/extensions/abilities/file-scan.js @@ -159,6 +159,11 @@ export function registerFileScan() { // Read-only — no confirmation needed. requiresConfirmation: false, + + // summarize() lists every plugin, theme and mu-plugin scanned plus the + // findings. That payload routinely exceeds maxToolResultLength, and a + // truncated security scan is worse than none. + preferSummarize: true, } ); } diff --git a/src/extensions/abilities/plugin-list.js b/src/extensions/abilities/plugin-list.js index 70314d6..00713a9 100644 --- a/src/extensions/abilities/plugin-list.js +++ b/src/extensions/abilities/plugin-list.js @@ -98,12 +98,12 @@ export function registerPluginList() { } are inactive.\n\n`; if ( activePlugins.length > 0 ) { - summary += ` * * Active plugins: * * ${ activePlugins.join( + summary += `**Active plugins:** ${ activePlugins.join( ', ' ) }\n\n`; } if ( inactivePlugins.length > 0 ) { - summary += ` * * Inactive plugins: * * ${ inactivePlugins.join( + summary += `**Inactive plugins:** ${ inactivePlugins.join( ', ' ) }`; } diff --git a/src/extensions/abilities/security-scan.js b/src/extensions/abilities/security-scan.js index 3631286..f92802c 100644 --- a/src/extensions/abilities/security-scan.js +++ b/src/extensions/abilities/security-scan.js @@ -95,6 +95,11 @@ export function registerSecurityScan() { }, requiresConfirmation: false, + + // summarize() renders the full pass/fail report grouped by severity. + // Re-rendering it through the LLM risks dropping findings when the + // payload exceeds maxToolResultLength. + preferSummarize: true, } ); } diff --git a/src/extensions/abilities/site-health.js b/src/extensions/abilities/site-health.js index 34250b6..98a80c1 100644 --- a/src/extensions/abilities/site-health.js +++ b/src/extensions/abilities/site-health.js @@ -85,15 +85,15 @@ export function registerSiteHealth() { // This creates a more natural conversational experience. if ( msg.includes( 'php' ) ) { - return `Your PHP version is * * ${ + return `Your PHP version is **${ result.php_version || 'Unknown' - } * * .`; + }**.`; } if ( msg.includes( 'wordpress' ) || msg.includes( 'wp version' ) ) { - return `You're running * * WordPress ${ + return `You're running **WordPress ${ result.wordpress_version || 'Unknown' - } * * .`; + }**.`; } if ( @@ -101,9 +101,9 @@ export function registerSiteHealth() { msg.includes( 'database' ) || msg.includes( 'db version' ) ) { - return `Your database is * * MySQL ${ + return `Your database is **MySQL ${ result.mysql_version || 'Unknown' - } * * .`; + }**.`; } if ( @@ -111,22 +111,22 @@ export function registerSiteHealth() { msg.includes( 'nginx' ) || msg.includes( 'apache' ) ) { - return `Your server is * * ${ + return `Your server is **${ result.server_software || 'Unknown' - } * * .`; + }**.`; } if ( msg.includes( 'theme' ) ) { // Handle nested object safely with optional chaining. - return `Your active theme is * * ${ + return `Your active theme is **${ result.active_theme?.name || 'Unknown' - } * * (version ${ result.active_theme?.version || '?' }).`; + }** (version ${ result.active_theme?.version || '?' }).`; } if ( msg.includes( 'memory' ) ) { - return `Your PHP memory limit is * * ${ + return `Your PHP memory limit is **${ result.memory_limit || 'Unknown' - } * * .`; + }**.`; } if ( @@ -134,27 +134,23 @@ export function registerSiteHealth() { msg.includes( 'site address' ) || msg.includes( 'home' ) ) { - return `Your site URL is * * ${ + return `Your site URL is **${ result.site_url || result.home_url || 'Unknown' - } * * .`; + }**.`; } // No specific question detected - return full health summary. // This is the default when user asks something general like "site health". return ( `Here's your site health information:\n\n` + - ` * * WordPress: * * ${ - result.wordpress_version || 'Unknown' - }\n` + - ` * * PHP: * * ${ result.php_version || 'Unknown' }\n` + - ` * * Database: * * MySQL ${ - result.mysql_version || 'Unknown' - }\n` + - ` * * Server: * * ${ result.server_software || 'Unknown' }\n` + - ` * * Theme: * * ${ result.active_theme?.name || 'Unknown' } (${ + `**WordPress:** ${ result.wordpress_version || 'Unknown' }\n` + + `**PHP:** ${ result.php_version || 'Unknown' }\n` + + `**Database:** MySQL ${ result.mysql_version || 'Unknown' }\n` + + `**Server:** ${ result.server_software || 'Unknown' }\n` + + `**Theme:** ${ result.active_theme?.name || 'Unknown' } (${ result.active_theme?.version || '?' })\n` + - ` * * Memory Limit: * * ${ result.memory_limit || 'Unknown' }` + `**Memory Limit:** ${ result.memory_limit || 'Unknown' }` ); }, @@ -209,6 +205,11 @@ export function registerSiteHealth() { // Read-only - no confirmation needed. requiresConfirmation: false, + + // summarize() already branches on the user's question (PHP version, theme, + // memory limit, ...) and returns a complete answer, so the second LLM call + // adds nothing but a truncation risk on the full health payload. + preferSummarize: true, } ); } diff --git a/src/extensions/abilities/verify-core-checksums.js b/src/extensions/abilities/verify-core-checksums.js index 561b1b1..146b8b1 100644 --- a/src/extensions/abilities/verify-core-checksums.js +++ b/src/extensions/abilities/verify-core-checksums.js @@ -191,6 +191,11 @@ export function registerVerifyCoreChecksums() { // Read-only - no confirmation needed. requiresConfirmation: false, + + // summarize() emits fenced diff blocks for modified core files. The LLM + // mangles fenced code and the diffs blow past maxToolResultLength, so + // render them directly. + preferSummarize: true, } ); } diff --git a/src/extensions/services/__tests__/react-agent.test.js b/src/extensions/services/__tests__/react-agent.test.js index 09e63e7..751223d 100644 --- a/src/extensions/services/__tests__/react-agent.test.js +++ b/src/extensions/services/__tests__/react-agent.test.js @@ -497,4 +497,93 @@ describe( 'ReactAgent', () => { expect( result.toolsUsed ).toContain( 'agentic-admin/site-health' ); } ); } ); + + describe( 'Structured output (grammar-constrained action envelope)', () => { + /** + * Returns the response_format of the Nth create() call, if any. + * + * @param {number} callIndex - Zero-based index of the create() call. + * @return {Object|undefined} The response_format passed, or undefined. + */ + function responseFormatOfCall( callIndex ) { + return mockEngine.chat.completions.create.mock.calls[ + callIndex + ]?.[ 0 ]?.response_format; + } + + it( 'should NOT constrain output while thinking is enabled', async () => { + // Thinking on is the default. A JSON grammar cannot represent the + // leading block, so the envelope must stay unconstrained. + mockStreamOnce( + mockEngine, + '{"action": "final_answer", "content": "Done"}' + ); + + await reactAgent.execute( 'hello', [] ); + + expect( responseFormatOfCall( 0 ) ).toBeUndefined(); + } ); + + it( 'should constrain output when thinking is disabled', async () => { + reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, { + disableThinking: true, + } ); + reactAgent.setCallbacks( mockCallbacks ); + + mockStreamOnce( + mockEngine, + '{"action": "final_answer", "content": "Done"}' + ); + + await reactAgent.execute( 'flush the cache', [] ); + + const format = responseFormatOfCall( 0 ); + expect( format ).toBeDefined(); + expect( format.type ).toBe( 'json_object' ); + + // The schema is passed stringified, as WebLLM expects. + const schema = JSON.parse( format.schema ); + expect( schema.required ).toEqual( [ 'action' ] ); + expect( schema.properties.action.enum ).toEqual( [ + 'tool_call', + 'final_answer', + ] ); + } ); + + it( 'should constrain follow-up turns once disableThinkingAfterTool fires', async () => { + reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, { + disableThinkingAfterTool: true, + } ); + reactAgent.setCallbacks( mockCallbacks ); + + mockStreamOnce( + mockEngine, + '{"action": "tool_call", "tool": "agentic-admin/plugin-list", "args": {}}', + '{"action": "final_answer", "content": "You have 1 plugin."}' + ); + + await reactAgent.execute( 'list plugins', [] ); + + // First turn still thinks, second turn (post-tool) does not. + expect( responseFormatOfCall( 0 ) ).toBeUndefined(); + expect( responseFormatOfCall( 1 ) ).toBeDefined(); + } ); + + it( 'should honour structuredOutput: false as an escape hatch', async () => { + reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, { + disableThinking: true, + structuredOutput: false, + } ); + reactAgent.setCallbacks( mockCallbacks ); + + mockStreamOnce( + mockEngine, + '{"action": "final_answer", "content": "Done"}' + ); + + await reactAgent.execute( 'flush the cache', [] ); + + expect( responseFormatOfCall( 0 ) ).toBeUndefined(); + } ); + } ); } ); diff --git a/src/extensions/services/external-engine.js b/src/extensions/services/external-engine.js index eed54a8..eedfdfa 100644 --- a/src/extensions/services/external-engine.js +++ b/src/extensions/services/external-engine.js @@ -153,6 +153,14 @@ class ExternalEngine { model: this.modelId, }; + // WebLLM accepts a `schema` key inside response_format to grammar-constrain + // decoding. That key is WebLLM-specific: OpenAI rejects unrecognized keys + // in response_format outright, and other providers ignore it. Keep plain + // JSON mode, which every OpenAI-compatible provider understands. + if ( body.response_format?.schema ) { + body.response_format = { type: body.response_format.type }; + } + // o1/o3/gpt-5 models require specific parameter mapping and don't support many standard params if ( isModernModel ) { if ( body.max_tokens ) { diff --git a/src/extensions/services/model-loader.js b/src/extensions/services/model-loader.js index d83e61d..f55e820 100644 --- a/src/extensions/services/model-loader.js +++ b/src/extensions/services/model-loader.js @@ -69,14 +69,6 @@ const log = createLogger( 'ModelLoader' ); */ const DEFAULT_MODEL = 'Qwen3-1.7B-q4f16_1-MLC'; -/** - * Model configuration options - */ -const MODEL_CONFIG = { - // Context window size (larger for 7B models) - context_window_size: 8192, -}; - /** * Mapping from f16 model IDs to their f32 equivalents. * Used when the GPU does not support the shader-f16 WebGPU feature. @@ -1376,7 +1368,6 @@ export { ModelLoader, modelLoader, DEFAULT_MODEL, - MODEL_CONFIG, MODEL_CONTEXT_SIZES, ExternalEngine, }; diff --git a/src/extensions/services/react-agent.js b/src/extensions/services/react-agent.js index 311081f..a5a4b04 100644 --- a/src/extensions/services/react-agent.js +++ b/src/extensions/services/react-agent.js @@ -49,6 +49,39 @@ const REACT_CONFIG = { maxToolResultLength: 3000, // 7B models handle more context disableThinking: false, // Disable Qwen 3 blocks for faster inference disableThinkingAfterTool: false, // Skip thinking on iterations after tool results + structuredOutput: true, // Grammar-constrain the action envelope (see REACT_ACTION_SCHEMA) +}; + +/** + * JSON schema for the ReAct action envelope. + * + * The loop only ever accepts two shapes: + * {"action": "tool_call", "tool": "", "args": {...}} + * {"action": "final_answer", "content": "..."} + * + * Passing this to the engine as `response_format` makes the grammar enforce + * the envelope during decoding, so malformed JSON becomes unrepresentable + * rather than something `parseActionFromResponse()` has to repair after the + * fact. WebLLM implements this in the WASM layer; OpenAI-compatible providers + * get plain JSON mode (see ExternalEngine, which strips the `schema` key). + * + * IMPORTANT: this can only be applied when thinking is disabled. Qwen 3 emits + * `...` *before* the JSON, which a JSON grammar forbids, so + * constraining a thinking turn would truncate the reasoning block and fail. + * The call site gates on `suppressThinkingUi` for exactly this reason. + */ +const REACT_ACTION_SCHEMA = { + type: 'object', + properties: { + action: { + type: 'string', + enum: [ 'tool_call', 'final_answer' ], + }, + tool: { type: 'string' }, + args: { type: 'object' }, + content: { type: 'string' }, + }, + required: [ 'action' ], }; /** @@ -255,6 +288,13 @@ class ReactAgent { ); try { + // Grammar-constrain the action envelope, but only on turns where + // thinking is off. A JSON grammar cannot represent the leading + // block, so constraining a thinking turn would break it. + // See REACT_ACTION_SCHEMA. + const useStructuredOutput = + this.config.structuredOutput && suppressThinkingUi; + // Stream LLM response to show thinking tokens live const stream = await engine.chat.completions.create( { messages, @@ -262,6 +302,12 @@ class ReactAgent { max_tokens: this.config.maxTokens, stream: true, stream_options: { include_usage: true }, + ...( useStructuredOutput && { + response_format: { + type: 'json_object', + schema: JSON.stringify( REACT_ACTION_SCHEMA ), + }, + } ), } ); let fullResponse = ''; From 9906d55177fe778f809239be49c92fbc96ffa450 Mon Sep 17 00:00:00 2001 From: pluginslab <57633278+pluginslab@users.noreply.github.com> Date: Sat, 1 Aug 2026 10:39:45 +0100 Subject: [PATCH 2/5] chore: log whether the action envelope was grammar-constrained Makes the structuredOutput gate observable in the browser console at DEBUG level, so the change can be validated manually against a real WebGPU engine. No behaviour change. Refs #229 --- src/extensions/services/react-agent.js | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/src/extensions/services/react-agent.js b/src/extensions/services/react-agent.js index a5a4b04..e4a79ae 100644 --- a/src/extensions/services/react-agent.js +++ b/src/extensions/services/react-agent.js @@ -295,6 +295,14 @@ class ReactAgent { const useStructuredOutput = this.config.structuredOutput && suppressThinkingUi; + log.debug( + `Structured output: ${ + useStructuredOutput + ? 'ON — action envelope is grammar-constrained' + : 'off — thinking turn, envelope unconstrained' + }` + ); + // Stream LLM response to show thinking tokens live const stream = await engine.chat.completions.create( { messages, From 0ca392fba5ee350dde70fb045faa601b93defd6b Mon Sep 17 00:00:00 2001 From: pluginslab <57633278+pluginslab@users.noreply.github.com> Date: Sat, 1 Aug 2026 10:47:04 +0100 Subject: [PATCH 3/5] fix: drop reserved wp prefix from the console log-level global window.wpAgenticLogLevel used the reserved `wp` prefix, the same class of issue the Plugins Team flagged on 28 May for the wpAgenticAdmin localize handle. The prefix rename in #227 missed it because it is set at runtime rather than declared in PHP. window.wpAgenticLogLevel -> window.agenticAdminLogLevel Also drops the redundant assignment before defineProperty, and clears the last "WP Agentic" branding strings from the SW console label and the SCSS headers, left over from the de-branding pass. Refs #228, #229 --- src/extensions/styles/admin-sidebar.scss | 2 +- src/extensions/styles/editor-sidebar.scss | 2 +- src/extensions/styles/main.scss | 2 +- src/extensions/sw.js | 2 +- src/extensions/utils/logger.js | 12 +++++++----- 5 files changed, 11 insertions(+), 9 deletions(-) diff --git a/src/extensions/styles/admin-sidebar.scss b/src/extensions/styles/admin-sidebar.scss index 752a64c..b6d1abf 100644 --- a/src/extensions/styles/admin-sidebar.scss +++ b/src/extensions/styles/admin-sidebar.scss @@ -1,7 +1,7 @@ /* stylelint-disable no-descending-specificity, no-duplicate-selectors */ /** - * WP Agentic Admin - Admin Sidebar Styles + * Agentic Admin - Admin Sidebar Styles * * Fixed sidebar panel for all wp-admin pages. * Imports base chat styles and adds sidebar-specific overrides. diff --git a/src/extensions/styles/editor-sidebar.scss b/src/extensions/styles/editor-sidebar.scss index 5f896dd..3086cf6 100644 --- a/src/extensions/styles/editor-sidebar.scss +++ b/src/extensions/styles/editor-sidebar.scss @@ -1,7 +1,7 @@ /* stylelint-disable no-descending-specificity, no-duplicate-selectors */ /** - * WP Agentic Admin - Editor Sidebar Styles + * Agentic Admin - Editor Sidebar Styles * * Imports base styles and adds overrides for the narrow * PluginSidebar panel (~280px width). diff --git a/src/extensions/styles/main.scss b/src/extensions/styles/main.scss index 1b1b249..346d88f 100644 --- a/src/extensions/styles/main.scss +++ b/src/extensions/styles/main.scss @@ -1,7 +1,7 @@ /* stylelint-disable no-descending-specificity, no-duplicate-selectors */ /** - * WP Agentic Admin - Main Styles + * Agentic Admin - Main Styles * * Descending specificity disabled: SCSS nesting of BEM modifiers causes * false positives when unrelated components style the same HTML elements. diff --git a/src/extensions/sw.js b/src/extensions/sw.js index 3d606bb..6b9c87d 100644 --- a/src/extensions/sw.js +++ b/src/extensions/sw.js @@ -30,7 +30,7 @@ const SW_VERSION = '0.4.101'; */ function swLog( ...args ) { const timestamp = new Date().toISOString().split( 'T' )[ 1 ].slice( 0, -1 ); - console.log( `[WP Agentic SW ${ timestamp }]`, ...args ); + console.log( `[Agentic Admin SW ${ timestamp }]`, ...args ); } /** diff --git a/src/extensions/utils/logger.js b/src/extensions/utils/logger.js index 32ef34a..290631d 100644 --- a/src/extensions/utils/logger.js +++ b/src/extensions/utils/logger.js @@ -15,19 +15,21 @@ const LOG_LEVELS = { DEBUG: 3, }; -// Set default log level (can be changed via console: window.wpAgenticLogLevel = 2) +// Set default log level (can be changed via console: window.agenticAdminLogLevel = 3) let currentLogLevel = LOG_LEVELS.INFO; -// Allow setting log level from browser console +// Allow setting log level from the browser console. +// Named without the reserved `wp` prefix, per the WordPress.org plugin +// guidelines on avoiding naming collisions. if ( typeof window !== 'undefined' ) { - window.wpAgenticLogLevel = currentLogLevel; - Object.defineProperty( window, 'wpAgenticLogLevel', { + Object.defineProperty( window, 'agenticAdminLogLevel', { + configurable: true, get: () => currentLogLevel, set: ( level ) => { if ( typeof level === 'number' && level >= 0 && level <= 3 ) { currentLogLevel = level; console.log( - `[WP Agentic] Log level set to: ${ Object.keys( + `[Agentic Admin] Log level set to: ${ Object.keys( LOG_LEVELS ).find( ( key ) => LOG_LEVELS[ key ] === level ) }` ); From 2ae01d9becc96c88a43759d3f3a5a1e978f46e10 Mon Sep 17 00:00:00 2001 From: pluginslab <57633278+pluginslab@users.noreply.github.com> Date: Sat, 1 Aug 2026 10:58:33 +0100 Subject: [PATCH 4/5] =?UTF-8?q?fix:=20default=20structuredOutput=20to=20of?= =?UTF-8?q?f=20=E2=80=94=20it=20breaks=20thinking=20models?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validated in a real browser (WebGPU, Service Worker mode) rather than against the mock, and the result contradicts the original assumption. Qwen2.5-7B-Instruct structuredOutput ON -> works, 2 iterations, 1 tool Qwen3-1.7B-q4f32_1 structuredOutput ON -> BROKEN, 0 tools, 2/2 runs On Qwen3 the model emits '{' followed by thousands of newlines until it hits max_tokens. parseActionFromResponse() then fails and the loop falls through to "I had trouble understanding how to help." Cause: Qwen 3 is a thinking model and wants to open before the JSON, which a JSON grammar cannot represent. /nothink is a soft instruction, so when the model still reaches for the grammar blocks every token except whitespace and decoding degenerates. Gating on suppressThinkingUi is necessary but not sufficient: it tracks whether we asked for no thinking, not whether the model complied. Since Qwen3-1.7B is DEFAULT_MODEL, shipping this on would regress the default install, so it becomes opt-in for non-thinking models. Machinery, schema and tests are kept. Tests now cover the off-by-default case explicitly. Refs #229 --- docs/ARCHITECTURE.md | 2 +- .../services/__tests__/react-agent.test.js | 50 +++++++++++-------- src/extensions/services/react-agent.js | 25 ++++++++-- 3 files changed, 50 insertions(+), 27 deletions(-) diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 838860d..426b120 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -326,7 +326,7 @@ The 1.7B model is recommended for most users — it loads faster, uses less VRAM - Repeated call detection (same tool twice = stop) - Max 10 iterations - Tool result truncation (`maxToolResultLength`, 3000 chars, in prompt-based mode) -- Schema-constrained JSON for the ReAct action envelope when thinking is disabled (see below) +- Schema-constrained JSON for the ReAct action envelope (`structuredOutput`, **off by default** — it breaks thinking models; see `REACT_ACTION_SCHEMA` in `react-agent.js`) - Context window overflow handling - JSON envelope unwrapping (prevents raw `{"action": "final_answer", ...}` leaking to user) diff --git a/src/extensions/services/__tests__/react-agent.test.js b/src/extensions/services/__tests__/react-agent.test.js index 751223d..cac53c8 100644 --- a/src/extensions/services/__tests__/react-agent.test.js +++ b/src/extensions/services/__tests__/react-agent.test.js @@ -511,9 +511,32 @@ describe( 'ReactAgent', () => { ]?.[ 0 ]?.response_format; } - it( 'should NOT constrain output while thinking is enabled', async () => { - // Thinking on is the default. A JSON grammar cannot represent the - // leading block, so the envelope must stay unconstrained. + it( 'should be OFF by default, even when thinking is disabled', async () => { + // Default is off because it breaks thinking models: Qwen 3 emits + // `{` then whitespace to max_tokens when the grammar blocks its + // block. Verified in-browser 2026-08-01, reproduced 2/2. + reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, { + disableThinking: true, + } ); + reactAgent.setCallbacks( mockCallbacks ); + + mockStreamOnce( + mockEngine, + '{"action": "final_answer", "content": "Done"}' + ); + + await reactAgent.execute( 'flush the cache', [] ); + + expect( responseFormatOfCall( 0 ) ).toBeUndefined(); + } ); + + it( 'should NOT constrain a thinking turn even when opted in', async () => { + // A JSON grammar cannot represent the leading block. + reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, { + structuredOutput: true, + } ); + reactAgent.setCallbacks( mockCallbacks ); + mockStreamOnce( mockEngine, '{"action": "final_answer", "content": "Done"}' @@ -524,8 +547,9 @@ describe( 'ReactAgent', () => { expect( responseFormatOfCall( 0 ) ).toBeUndefined(); } ); - it( 'should constrain output when thinking is disabled', async () => { + it( 'should constrain output when opted in AND thinking is disabled', async () => { reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, { + structuredOutput: true, disableThinking: true, } ); reactAgent.setCallbacks( mockCallbacks ); @@ -552,6 +576,7 @@ describe( 'ReactAgent', () => { it( 'should constrain follow-up turns once disableThinkingAfterTool fires', async () => { reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, { + structuredOutput: true, disableThinkingAfterTool: true, } ); reactAgent.setCallbacks( mockCallbacks ); @@ -568,22 +593,5 @@ describe( 'ReactAgent', () => { expect( responseFormatOfCall( 0 ) ).toBeUndefined(); expect( responseFormatOfCall( 1 ) ).toBeDefined(); } ); - - it( 'should honour structuredOutput: false as an escape hatch', async () => { - reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, { - disableThinking: true, - structuredOutput: false, - } ); - reactAgent.setCallbacks( mockCallbacks ); - - mockStreamOnce( - mockEngine, - '{"action": "final_answer", "content": "Done"}' - ); - - await reactAgent.execute( 'flush the cache', [] ); - - expect( responseFormatOfCall( 0 ) ).toBeUndefined(); - } ); } ); } ); diff --git a/src/extensions/services/react-agent.js b/src/extensions/services/react-agent.js index e4a79ae..8087e9c 100644 --- a/src/extensions/services/react-agent.js +++ b/src/extensions/services/react-agent.js @@ -49,7 +49,7 @@ const REACT_CONFIG = { maxToolResultLength: 3000, // 7B models handle more context disableThinking: false, // Disable Qwen 3 blocks for faster inference disableThinkingAfterTool: false, // Skip thinking on iterations after tool results - structuredOutput: true, // Grammar-constrain the action envelope (see REACT_ACTION_SCHEMA) + structuredOutput: false, // Opt-in. Breaks thinking models — see REACT_ACTION_SCHEMA }; /** @@ -65,10 +65,25 @@ const REACT_CONFIG = { * fact. WebLLM implements this in the WASM layer; OpenAI-compatible providers * get plain JSON mode (see ExternalEngine, which strips the `schema` key). * - * IMPORTANT: this can only be applied when thinking is disabled. Qwen 3 emits - * `...` *before* the JSON, which a JSON grammar forbids, so - * constraining a thinking turn would truncate the reasoning block and fail. - * The call site gates on `suppressThinkingUi` for exactly this reason. + * OFF BY DEFAULT — it breaks thinking models. Measured in a real browser + * (WebGPU, Service Worker mode) on 2026-08-01: + * + * Qwen2.5-7B-Instruct structuredOutput ON -> works, 2 iterations, 1 tool + * Qwen3-1.7B-q4f32_1 structuredOutput ON -> BROKEN, 0 tools, reproduced 2/2 + * + * On Qwen3 the model emits `{` and then thousands of newlines until it hits + * max_tokens, so `parseActionFromResponse()` fails and the loop falls through + * to "I had trouble understanding how to help." + * + * Cause: Qwen 3 is a thinking model and wants to open `` before the + * JSON. A JSON grammar cannot represent that block. `/nothink` is only a soft + * instruction, so when the model still reaches for `` the grammar + * blocks every token except whitespace and decoding degenerates. + * + * Gating on `suppressThinkingUi` (below) is therefore necessary but NOT + * sufficient: it tracks whether we *asked* for no thinking, not whether the + * model complied. Enable this only on a non-thinking model such as + * Qwen2.5-7B, via `new ReactAgent( ..., { structuredOutput: true } )`. */ const REACT_ACTION_SCHEMA = { type: 'object', From a793075387d52782728570f7ae1fa94f8e35fdd8 Mon Sep 17 00:00:00 2001 From: pluginslab <57633278+pluginslab@users.noreply.github.com> Date: Sat, 1 Aug 2026 11:02:33 +0100 Subject: [PATCH 5/5] chore: log the real reason structured output was skipped The log said "thinking turn" whenever the constraint was absent, but with structuredOutput now defaulting to off that is the wrong cause and it misled during in-browser debugging. Distinguish disabled-by-config from thinking-turn. Refs #229 --- src/extensions/services/react-agent.js | 21 ++++++++++++++------- 1 file changed, 14 insertions(+), 7 deletions(-) diff --git a/src/extensions/services/react-agent.js b/src/extensions/services/react-agent.js index 8087e9c..f61b774 100644 --- a/src/extensions/services/react-agent.js +++ b/src/extensions/services/react-agent.js @@ -310,13 +310,20 @@ class ReactAgent { const useStructuredOutput = this.config.structuredOutput && suppressThinkingUi; - log.debug( - `Structured output: ${ - useStructuredOutput - ? 'ON — action envelope is grammar-constrained' - : 'off — thinking turn, envelope unconstrained' - }` - ); + // Name the actual reason — "disabled" and "thinking turn" are + // different causes and conflating them misleads when debugging. + let structuredOutputReason; + if ( useStructuredOutput ) { + structuredOutputReason = + 'ON — action envelope is grammar-constrained'; + } else if ( ! this.config.structuredOutput ) { + structuredOutputReason = + 'off — structuredOutput disabled (default)'; + } else { + structuredOutputReason = + 'off — thinking turn, envelope unconstrained'; + } + log.debug( `Structured output: ${ structuredOutputReason }` ); // Stream LLM response to show thinking tokens live const stream = await engine.chat.completions.create( {