diff --git a/.claude/CLAUDE.md b/.claude/CLAUDE.md index 25a76e1..cb78d69 100644 --- a/.claude/CLAUDE.md +++ b/.claude/CLAUDE.md @@ -58,8 +58,8 @@ Update ALL of these: ## Constraints -- **Context window**: 4096 tokens (1.7B) / 32768 (7B). Tool descriptions sent every request — keep them concise. -- **Tool results truncated** to 2000 chars before sending to LLM. +- **Context window**: 8192 tokens (1.7B) / 32768 (7B), 8192 default. Source of truth is `MODEL_CONTEXT_SIZES` in `model-loader.js`; override per model via the `agentic_admin_context_size` localStorage key. Tool descriptions sent every request — keep them concise. +- **Tool results truncated** to `maxToolResultLength` (3000 chars) before sending to LLM. Abilities with `preferSummarize: true` bypass this entirely. - **Max 10 ReAct iterations**. Repeated tool call detection stops oscillation. - **Service Worker** (`sw.js`) must be self-contained — no code splitting, no dynamic imports. diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index e40fe3b..426b120 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -319,13 +319,14 @@ The 1.7B model is recommended for most users — it loads faster, uses less VRAM 1. **JSON formatting** - 7B models produce valid JSON consistently (100% parse success with Qwen2.5-7B). Robust parsing still handles edge cases: try native `JSON.parse` first, then fall back to quote sanitization. 2. **Goal efficiency** - Qwen2.5-7B calls exactly 1 tool for single-goal tasks (no over-shooting). 3. **Multi-step reasoning** - 100% success on conditional logic and diagnose-then-fix chains. -4. **Context limits** - 4096 token context window configured (models support up to 32K). +4. **Context limits** - Context window is set per model in `MODEL_CONTEXT_SIZES` (`model-loader.js`): 8192 tokens for Qwen3 1.7B, 32768 for Qwen2.5 7B, 8192 default fallback. A per-model override can be stored in the `agentic_admin_context_size` localStorage key and is resolved by `ModelLoader.getEffectiveContextSize()`. ### Safety Mechanisms - Repeated call detection (same tool twice = stop) - Max 10 iterations -- Tool result truncation (2000 chars max in prompt-based mode) +- Tool result truncation (`maxToolResultLength`, 3000 chars, in prompt-based mode) +- Schema-constrained JSON for the ReAct action envelope (`structuredOutput`, **off by default** — it breaks thinking models; see `REACT_ACTION_SCHEMA` in `react-agent.js`) - Context window overflow handling - JSON envelope unwrapping (prevents raw `{"action": "final_answer", ...}` leaking to user) diff --git a/src/extensions/abilities/core-environment-info.js b/src/extensions/abilities/core-environment-info.js index 5bdd609..47f9ff6 100644 --- a/src/extensions/abilities/core-environment-info.js +++ b/src/extensions/abilities/core-environment-info.js @@ -71,18 +71,16 @@ export function registerCoreEnvironmentInfo() { const envDisplay = result.environment.charAt( 0 ).toUpperCase() + result.environment.slice( 1 ); - lines.push( ` * * Environment: * * ${ envDisplay }` ); + lines.push( `**Environment:** ${ envDisplay }` ); } if ( result.wp_version ) { - lines.push( - ` * * WordPress Version: * * ${ result.wp_version }` - ); + lines.push( `**WordPress Version:** ${ result.wp_version }` ); } if ( result.php_version ) { - lines.push( ` * * PHP Version: * * ${ result.php_version }` ); + lines.push( `**PHP Version:** ${ result.php_version }` ); } if ( result.db_server_info ) { - lines.push( ` * * Database: * * ${ result.db_server_info }` ); + lines.push( `**Database:** ${ result.db_server_info }` ); } if ( lines.length === 0 ) { diff --git a/src/extensions/abilities/core-site-info.js b/src/extensions/abilities/core-site-info.js index 03801e4..042f323 100644 --- a/src/extensions/abilities/core-site-info.js +++ b/src/extensions/abilities/core-site-info.js @@ -110,25 +110,25 @@ export function registerCoreSiteInfo() { const lines = []; if ( result.name ) { - lines.push( ` * * Site Name: * * ${ result.name }` ); + lines.push( `**Site Name:** ${ result.name }` ); } if ( result.description ) { - lines.push( ` * * Tagline: * * ${ result.description }` ); + lines.push( `**Tagline:** ${ result.description }` ); } if ( result.url ) { - lines.push( ` * * Site URL: * * ${ result.url }` ); + lines.push( `**Site URL:** ${ result.url }` ); } if ( result.version ) { - lines.push( ` * * WordPress Version: * * ${ result.version }` ); + lines.push( `**WordPress Version:** ${ result.version }` ); } if ( result.language ) { - lines.push( ` * * Language: * * ${ result.language }` ); + lines.push( `**Language:** ${ result.language }` ); } if ( result.admin_email ) { - lines.push( ` * * Admin Email: * * ${ result.admin_email }` ); + lines.push( `**Admin Email:** ${ result.admin_email }` ); } if ( result.charset ) { - lines.push( ` * * Charset: * * ${ result.charset }` ); + lines.push( `**Charset:** ${ result.charset }` ); } if ( lines.length === 0 ) { diff --git a/src/extensions/abilities/file-scan.js b/src/extensions/abilities/file-scan.js index c3390b6..3a55e74 100644 --- a/src/extensions/abilities/file-scan.js +++ b/src/extensions/abilities/file-scan.js @@ -159,6 +159,11 @@ export function registerFileScan() { // Read-only — no confirmation needed. requiresConfirmation: false, + + // summarize() lists every plugin, theme and mu-plugin scanned plus the + // findings. That payload routinely exceeds maxToolResultLength, and a + // truncated security scan is worse than none. + preferSummarize: true, } ); } diff --git a/src/extensions/abilities/plugin-list.js b/src/extensions/abilities/plugin-list.js index 70314d6..00713a9 100644 --- a/src/extensions/abilities/plugin-list.js +++ b/src/extensions/abilities/plugin-list.js @@ -98,12 +98,12 @@ export function registerPluginList() { } are inactive.\n\n`; if ( activePlugins.length > 0 ) { - summary += ` * * Active plugins: * * ${ activePlugins.join( + summary += `**Active plugins:** ${ activePlugins.join( ', ' ) }\n\n`; } if ( inactivePlugins.length > 0 ) { - summary += ` * * Inactive plugins: * * ${ inactivePlugins.join( + summary += `**Inactive plugins:** ${ inactivePlugins.join( ', ' ) }`; } diff --git a/src/extensions/abilities/security-scan.js b/src/extensions/abilities/security-scan.js index 3631286..f92802c 100644 --- a/src/extensions/abilities/security-scan.js +++ b/src/extensions/abilities/security-scan.js @@ -95,6 +95,11 @@ export function registerSecurityScan() { }, requiresConfirmation: false, + + // summarize() renders the full pass/fail report grouped by severity. + // Re-rendering it through the LLM risks dropping findings when the + // payload exceeds maxToolResultLength. + preferSummarize: true, } ); } diff --git a/src/extensions/abilities/site-health.js b/src/extensions/abilities/site-health.js index 34250b6..98a80c1 100644 --- a/src/extensions/abilities/site-health.js +++ b/src/extensions/abilities/site-health.js @@ -85,15 +85,15 @@ export function registerSiteHealth() { // This creates a more natural conversational experience. if ( msg.includes( 'php' ) ) { - return `Your PHP version is * * ${ + return `Your PHP version is **${ result.php_version || 'Unknown' - } * * .`; + }**.`; } if ( msg.includes( 'wordpress' ) || msg.includes( 'wp version' ) ) { - return `You're running * * WordPress ${ + return `You're running **WordPress ${ result.wordpress_version || 'Unknown' - } * * .`; + }**.`; } if ( @@ -101,9 +101,9 @@ export function registerSiteHealth() { msg.includes( 'database' ) || msg.includes( 'db version' ) ) { - return `Your database is * * MySQL ${ + return `Your database is **MySQL ${ result.mysql_version || 'Unknown' - } * * .`; + }**.`; } if ( @@ -111,22 +111,22 @@ export function registerSiteHealth() { msg.includes( 'nginx' ) || msg.includes( 'apache' ) ) { - return `Your server is * * ${ + return `Your server is **${ result.server_software || 'Unknown' - } * * .`; + }**.`; } if ( msg.includes( 'theme' ) ) { // Handle nested object safely with optional chaining. - return `Your active theme is * * ${ + return `Your active theme is **${ result.active_theme?.name || 'Unknown' - } * * (version ${ result.active_theme?.version || '?' }).`; + }** (version ${ result.active_theme?.version || '?' }).`; } if ( msg.includes( 'memory' ) ) { - return `Your PHP memory limit is * * ${ + return `Your PHP memory limit is **${ result.memory_limit || 'Unknown' - } * * .`; + }**.`; } if ( @@ -134,27 +134,23 @@ export function registerSiteHealth() { msg.includes( 'site address' ) || msg.includes( 'home' ) ) { - return `Your site URL is * * ${ + return `Your site URL is **${ result.site_url || result.home_url || 'Unknown' - } * * .`; + }**.`; } // No specific question detected - return full health summary. // This is the default when user asks something general like "site health". return ( `Here's your site health information:\n\n` + - ` * * WordPress: * * ${ - result.wordpress_version || 'Unknown' - }\n` + - ` * * PHP: * * ${ result.php_version || 'Unknown' }\n` + - ` * * Database: * * MySQL ${ - result.mysql_version || 'Unknown' - }\n` + - ` * * Server: * * ${ result.server_software || 'Unknown' }\n` + - ` * * Theme: * * ${ result.active_theme?.name || 'Unknown' } (${ + `**WordPress:** ${ result.wordpress_version || 'Unknown' }\n` + + `**PHP:** ${ result.php_version || 'Unknown' }\n` + + `**Database:** MySQL ${ result.mysql_version || 'Unknown' }\n` + + `**Server:** ${ result.server_software || 'Unknown' }\n` + + `**Theme:** ${ result.active_theme?.name || 'Unknown' } (${ result.active_theme?.version || '?' })\n` + - ` * * Memory Limit: * * ${ result.memory_limit || 'Unknown' }` + `**Memory Limit:** ${ result.memory_limit || 'Unknown' }` ); }, @@ -209,6 +205,11 @@ export function registerSiteHealth() { // Read-only - no confirmation needed. requiresConfirmation: false, + + // summarize() already branches on the user's question (PHP version, theme, + // memory limit, ...) and returns a complete answer, so the second LLM call + // adds nothing but a truncation risk on the full health payload. + preferSummarize: true, } ); } diff --git a/src/extensions/abilities/verify-core-checksums.js b/src/extensions/abilities/verify-core-checksums.js index 561b1b1..146b8b1 100644 --- a/src/extensions/abilities/verify-core-checksums.js +++ b/src/extensions/abilities/verify-core-checksums.js @@ -191,6 +191,11 @@ export function registerVerifyCoreChecksums() { // Read-only - no confirmation needed. requiresConfirmation: false, + + // summarize() emits fenced diff blocks for modified core files. The LLM + // mangles fenced code and the diffs blow past maxToolResultLength, so + // render them directly. + preferSummarize: true, } ); } diff --git a/src/extensions/services/__tests__/react-agent.test.js b/src/extensions/services/__tests__/react-agent.test.js index 09e63e7..cac53c8 100644 --- a/src/extensions/services/__tests__/react-agent.test.js +++ b/src/extensions/services/__tests__/react-agent.test.js @@ -497,4 +497,101 @@ describe( 'ReactAgent', () => { expect( result.toolsUsed ).toContain( 'agentic-admin/site-health' ); } ); } ); + + describe( 'Structured output (grammar-constrained action envelope)', () => { + /** + * Returns the response_format of the Nth create() call, if any. + * + * @param {number} callIndex - Zero-based index of the create() call. + * @return {Object|undefined} The response_format passed, or undefined. + */ + function responseFormatOfCall( callIndex ) { + return mockEngine.chat.completions.create.mock.calls[ + callIndex + ]?.[ 0 ]?.response_format; + } + + it( 'should be OFF by default, even when thinking is disabled', async () => { + // Default is off because it breaks thinking models: Qwen 3 emits + // `{` then whitespace to max_tokens when the grammar blocks its + // block. Verified in-browser 2026-08-01, reproduced 2/2. + reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, { + disableThinking: true, + } ); + reactAgent.setCallbacks( mockCallbacks ); + + mockStreamOnce( + mockEngine, + '{"action": "final_answer", "content": "Done"}' + ); + + await reactAgent.execute( 'flush the cache', [] ); + + expect( responseFormatOfCall( 0 ) ).toBeUndefined(); + } ); + + it( 'should NOT constrain a thinking turn even when opted in', async () => { + // A JSON grammar cannot represent the leading block. + reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, { + structuredOutput: true, + } ); + reactAgent.setCallbacks( mockCallbacks ); + + mockStreamOnce( + mockEngine, + '{"action": "final_answer", "content": "Done"}' + ); + + await reactAgent.execute( 'hello', [] ); + + expect( responseFormatOfCall( 0 ) ).toBeUndefined(); + } ); + + it( 'should constrain output when opted in AND thinking is disabled', async () => { + reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, { + structuredOutput: true, + disableThinking: true, + } ); + reactAgent.setCallbacks( mockCallbacks ); + + mockStreamOnce( + mockEngine, + '{"action": "final_answer", "content": "Done"}' + ); + + await reactAgent.execute( 'flush the cache', [] ); + + const format = responseFormatOfCall( 0 ); + expect( format ).toBeDefined(); + expect( format.type ).toBe( 'json_object' ); + + // The schema is passed stringified, as WebLLM expects. + const schema = JSON.parse( format.schema ); + expect( schema.required ).toEqual( [ 'action' ] ); + expect( schema.properties.action.enum ).toEqual( [ + 'tool_call', + 'final_answer', + ] ); + } ); + + it( 'should constrain follow-up turns once disableThinkingAfterTool fires', async () => { + reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, { + structuredOutput: true, + disableThinkingAfterTool: true, + } ); + reactAgent.setCallbacks( mockCallbacks ); + + mockStreamOnce( + mockEngine, + '{"action": "tool_call", "tool": "agentic-admin/plugin-list", "args": {}}', + '{"action": "final_answer", "content": "You have 1 plugin."}' + ); + + await reactAgent.execute( 'list plugins', [] ); + + // First turn still thinks, second turn (post-tool) does not. + expect( responseFormatOfCall( 0 ) ).toBeUndefined(); + expect( responseFormatOfCall( 1 ) ).toBeDefined(); + } ); + } ); } ); diff --git a/src/extensions/services/external-engine.js b/src/extensions/services/external-engine.js index eed54a8..eedfdfa 100644 --- a/src/extensions/services/external-engine.js +++ b/src/extensions/services/external-engine.js @@ -153,6 +153,14 @@ class ExternalEngine { model: this.modelId, }; + // WebLLM accepts a `schema` key inside response_format to grammar-constrain + // decoding. That key is WebLLM-specific: OpenAI rejects unrecognized keys + // in response_format outright, and other providers ignore it. Keep plain + // JSON mode, which every OpenAI-compatible provider understands. + if ( body.response_format?.schema ) { + body.response_format = { type: body.response_format.type }; + } + // o1/o3/gpt-5 models require specific parameter mapping and don't support many standard params if ( isModernModel ) { if ( body.max_tokens ) { diff --git a/src/extensions/services/model-loader.js b/src/extensions/services/model-loader.js index d83e61d..f55e820 100644 --- a/src/extensions/services/model-loader.js +++ b/src/extensions/services/model-loader.js @@ -69,14 +69,6 @@ const log = createLogger( 'ModelLoader' ); */ const DEFAULT_MODEL = 'Qwen3-1.7B-q4f16_1-MLC'; -/** - * Model configuration options - */ -const MODEL_CONFIG = { - // Context window size (larger for 7B models) - context_window_size: 8192, -}; - /** * Mapping from f16 model IDs to their f32 equivalents. * Used when the GPU does not support the shader-f16 WebGPU feature. @@ -1376,7 +1368,6 @@ export { ModelLoader, modelLoader, DEFAULT_MODEL, - MODEL_CONFIG, MODEL_CONTEXT_SIZES, ExternalEngine, }; diff --git a/src/extensions/services/react-agent.js b/src/extensions/services/react-agent.js index 311081f..f61b774 100644 --- a/src/extensions/services/react-agent.js +++ b/src/extensions/services/react-agent.js @@ -49,6 +49,54 @@ const REACT_CONFIG = { maxToolResultLength: 3000, // 7B models handle more context disableThinking: false, // Disable Qwen 3 blocks for faster inference disableThinkingAfterTool: false, // Skip thinking on iterations after tool results + structuredOutput: false, // Opt-in. Breaks thinking models — see REACT_ACTION_SCHEMA +}; + +/** + * JSON schema for the ReAct action envelope. + * + * The loop only ever accepts two shapes: + * {"action": "tool_call", "tool": "", "args": {...}} + * {"action": "final_answer", "content": "..."} + * + * Passing this to the engine as `response_format` makes the grammar enforce + * the envelope during decoding, so malformed JSON becomes unrepresentable + * rather than something `parseActionFromResponse()` has to repair after the + * fact. WebLLM implements this in the WASM layer; OpenAI-compatible providers + * get plain JSON mode (see ExternalEngine, which strips the `schema` key). + * + * OFF BY DEFAULT — it breaks thinking models. Measured in a real browser + * (WebGPU, Service Worker mode) on 2026-08-01: + * + * Qwen2.5-7B-Instruct structuredOutput ON -> works, 2 iterations, 1 tool + * Qwen3-1.7B-q4f32_1 structuredOutput ON -> BROKEN, 0 tools, reproduced 2/2 + * + * On Qwen3 the model emits `{` and then thousands of newlines until it hits + * max_tokens, so `parseActionFromResponse()` fails and the loop falls through + * to "I had trouble understanding how to help." + * + * Cause: Qwen 3 is a thinking model and wants to open `` before the + * JSON. A JSON grammar cannot represent that block. `/nothink` is only a soft + * instruction, so when the model still reaches for `` the grammar + * blocks every token except whitespace and decoding degenerates. + * + * Gating on `suppressThinkingUi` (below) is therefore necessary but NOT + * sufficient: it tracks whether we *asked* for no thinking, not whether the + * model complied. Enable this only on a non-thinking model such as + * Qwen2.5-7B, via `new ReactAgent( ..., { structuredOutput: true } )`. + */ +const REACT_ACTION_SCHEMA = { + type: 'object', + properties: { + action: { + type: 'string', + enum: [ 'tool_call', 'final_answer' ], + }, + tool: { type: 'string' }, + args: { type: 'object' }, + content: { type: 'string' }, + }, + required: [ 'action' ], }; /** @@ -255,6 +303,28 @@ class ReactAgent { ); try { + // Grammar-constrain the action envelope, but only on turns where + // thinking is off. A JSON grammar cannot represent the leading + // block, so constraining a thinking turn would break it. + // See REACT_ACTION_SCHEMA. + const useStructuredOutput = + this.config.structuredOutput && suppressThinkingUi; + + // Name the actual reason — "disabled" and "thinking turn" are + // different causes and conflating them misleads when debugging. + let structuredOutputReason; + if ( useStructuredOutput ) { + structuredOutputReason = + 'ON — action envelope is grammar-constrained'; + } else if ( ! this.config.structuredOutput ) { + structuredOutputReason = + 'off — structuredOutput disabled (default)'; + } else { + structuredOutputReason = + 'off — thinking turn, envelope unconstrained'; + } + log.debug( `Structured output: ${ structuredOutputReason }` ); + // Stream LLM response to show thinking tokens live const stream = await engine.chat.completions.create( { messages, @@ -262,6 +332,12 @@ class ReactAgent { max_tokens: this.config.maxTokens, stream: true, stream_options: { include_usage: true }, + ...( useStructuredOutput && { + response_format: { + type: 'json_object', + schema: JSON.stringify( REACT_ACTION_SCHEMA ), + }, + } ), } ); let fullResponse = ''; diff --git a/src/extensions/styles/admin-sidebar.scss b/src/extensions/styles/admin-sidebar.scss index 752a64c..b6d1abf 100644 --- a/src/extensions/styles/admin-sidebar.scss +++ b/src/extensions/styles/admin-sidebar.scss @@ -1,7 +1,7 @@ /* stylelint-disable no-descending-specificity, no-duplicate-selectors */ /** - * WP Agentic Admin - Admin Sidebar Styles + * Agentic Admin - Admin Sidebar Styles * * Fixed sidebar panel for all wp-admin pages. * Imports base chat styles and adds sidebar-specific overrides. diff --git a/src/extensions/styles/editor-sidebar.scss b/src/extensions/styles/editor-sidebar.scss index 5f896dd..3086cf6 100644 --- a/src/extensions/styles/editor-sidebar.scss +++ b/src/extensions/styles/editor-sidebar.scss @@ -1,7 +1,7 @@ /* stylelint-disable no-descending-specificity, no-duplicate-selectors */ /** - * WP Agentic Admin - Editor Sidebar Styles + * Agentic Admin - Editor Sidebar Styles * * Imports base styles and adds overrides for the narrow * PluginSidebar panel (~280px width). diff --git a/src/extensions/styles/main.scss b/src/extensions/styles/main.scss index 1b1b249..346d88f 100644 --- a/src/extensions/styles/main.scss +++ b/src/extensions/styles/main.scss @@ -1,7 +1,7 @@ /* stylelint-disable no-descending-specificity, no-duplicate-selectors */ /** - * WP Agentic Admin - Main Styles + * Agentic Admin - Main Styles * * Descending specificity disabled: SCSS nesting of BEM modifiers causes * false positives when unrelated components style the same HTML elements. diff --git a/src/extensions/sw.js b/src/extensions/sw.js index 3d606bb..6b9c87d 100644 --- a/src/extensions/sw.js +++ b/src/extensions/sw.js @@ -30,7 +30,7 @@ const SW_VERSION = '0.4.101'; */ function swLog( ...args ) { const timestamp = new Date().toISOString().split( 'T' )[ 1 ].slice( 0, -1 ); - console.log( `[WP Agentic SW ${ timestamp }]`, ...args ); + console.log( `[Agentic Admin SW ${ timestamp }]`, ...args ); } /** diff --git a/src/extensions/utils/logger.js b/src/extensions/utils/logger.js index 32ef34a..290631d 100644 --- a/src/extensions/utils/logger.js +++ b/src/extensions/utils/logger.js @@ -15,19 +15,21 @@ const LOG_LEVELS = { DEBUG: 3, }; -// Set default log level (can be changed via console: window.wpAgenticLogLevel = 2) +// Set default log level (can be changed via console: window.agenticAdminLogLevel = 3) let currentLogLevel = LOG_LEVELS.INFO; -// Allow setting log level from browser console +// Allow setting log level from the browser console. +// Named without the reserved `wp` prefix, per the WordPress.org plugin +// guidelines on avoiding naming collisions. if ( typeof window !== 'undefined' ) { - window.wpAgenticLogLevel = currentLogLevel; - Object.defineProperty( window, 'wpAgenticLogLevel', { + Object.defineProperty( window, 'agenticAdminLogLevel', { + configurable: true, get: () => currentLogLevel, set: ( level ) => { if ( typeof level === 'number' && level >= 0 && level <= 3 ) { currentLogLevel = level; console.log( - `[WP Agentic] Log level set to: ${ Object.keys( + `[Agentic Admin] Log level set to: ${ Object.keys( LOG_LEVELS ).find( ( key ) => LOG_LEVELS[ key ] === level ) }` );