Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions .claude/CLAUDE.md
Original file line number Diff line number Diff line change
Expand Up @@ -58,8 +58,8 @@ Update ALL of these:

## Constraints

- **Context window**: 4096 tokens (1.7B) / 32768 (7B). Tool descriptions sent every request — keep them concise.
- **Tool results truncated** to 2000 chars before sending to LLM.
- **Context window**: 8192 tokens (1.7B) / 32768 (7B), 8192 default. Source of truth is `MODEL_CONTEXT_SIZES` in `model-loader.js`; override per model via the `agentic_admin_context_size` localStorage key. Tool descriptions sent every request — keep them concise.
- **Tool results truncated** to `maxToolResultLength` (3000 chars) before sending to LLM. Abilities with `preferSummarize: true` bypass this entirely.
- **Max 10 ReAct iterations**. Repeated tool call detection stops oscillation.
- **Service Worker** (`sw.js`) must be self-contained — no code splitting, no dynamic imports.

Expand Down
5 changes: 3 additions & 2 deletions docs/ARCHITECTURE.md
Original file line number Diff line number Diff line change
Expand Up @@ -319,13 +319,14 @@ The 1.7B model is recommended for most users — it loads faster, uses less VRAM
1. **JSON formatting** - 7B models produce valid JSON consistently (100% parse success with Qwen2.5-7B). Robust parsing still handles edge cases: try native `JSON.parse` first, then fall back to quote sanitization.
2. **Goal efficiency** - Qwen2.5-7B calls exactly 1 tool for single-goal tasks (no over-shooting).
3. **Multi-step reasoning** - 100% success on conditional logic and diagnose-then-fix chains.
4. **Context limits** - 4096 token context window configured (models support up to 32K).
4. **Context limits** - Context window is set per model in `MODEL_CONTEXT_SIZES` (`model-loader.js`): 8192 tokens for Qwen3 1.7B, 32768 for Qwen2.5 7B, 8192 default fallback. A per-model override can be stored in the `agentic_admin_context_size` localStorage key and is resolved by `ModelLoader.getEffectiveContextSize()`.

### Safety Mechanisms

- Repeated call detection (same tool twice = stop)
- Max 10 iterations
- Tool result truncation (2000 chars max in prompt-based mode)
- Tool result truncation (`maxToolResultLength`, 3000 chars, in prompt-based mode)
- Schema-constrained JSON for the ReAct action envelope (`structuredOutput`, **off by default** — it breaks thinking models; see `REACT_ACTION_SCHEMA` in `react-agent.js`)
- Context window overflow handling
- JSON envelope unwrapping (prevents raw `{"action": "final_answer", ...}` leaking to user)

Expand Down
10 changes: 4 additions & 6 deletions src/extensions/abilities/core-environment-info.js
Original file line number Diff line number Diff line change
Expand Up @@ -71,18 +71,16 @@ export function registerCoreEnvironmentInfo() {
const envDisplay =
result.environment.charAt( 0 ).toUpperCase() +
result.environment.slice( 1 );
lines.push( ` * * Environment: * * ${ envDisplay }` );
lines.push( `**Environment:** ${ envDisplay }` );
}
if ( result.wp_version ) {
lines.push(
` * * WordPress Version: * * ${ result.wp_version }`
);
lines.push( `**WordPress Version:** ${ result.wp_version }` );
}
if ( result.php_version ) {
lines.push( ` * * PHP Version: * * ${ result.php_version }` );
lines.push( `**PHP Version:** ${ result.php_version }` );
}
if ( result.db_server_info ) {
lines.push( ` * * Database: * * ${ result.db_server_info }` );
lines.push( `**Database:** ${ result.db_server_info }` );
}

if ( lines.length === 0 ) {
Expand Down
14 changes: 7 additions & 7 deletions src/extensions/abilities/core-site-info.js
Original file line number Diff line number Diff line change
Expand Up @@ -110,25 +110,25 @@ export function registerCoreSiteInfo() {
const lines = [];

if ( result.name ) {
lines.push( ` * * Site Name: * * ${ result.name }` );
lines.push( `**Site Name:** ${ result.name }` );
}
if ( result.description ) {
lines.push( ` * * Tagline: * * ${ result.description }` );
lines.push( `**Tagline:** ${ result.description }` );
}
if ( result.url ) {
lines.push( ` * * Site URL: * * ${ result.url }` );
lines.push( `**Site URL:** ${ result.url }` );
}
if ( result.version ) {
lines.push( ` * * WordPress Version: * * ${ result.version }` );
lines.push( `**WordPress Version:** ${ result.version }` );
}
if ( result.language ) {
lines.push( ` * * Language: * * ${ result.language }` );
lines.push( `**Language:** ${ result.language }` );
}
if ( result.admin_email ) {
lines.push( ` * * Admin Email: * * ${ result.admin_email }` );
lines.push( `**Admin Email:** ${ result.admin_email }` );
}
if ( result.charset ) {
lines.push( ` * * Charset: * * ${ result.charset }` );
lines.push( `**Charset:** ${ result.charset }` );
}

if ( lines.length === 0 ) {
Expand Down
5 changes: 5 additions & 0 deletions src/extensions/abilities/file-scan.js
Original file line number Diff line number Diff line change
Expand Up @@ -159,6 +159,11 @@ export function registerFileScan() {

// Read-only — no confirmation needed.
requiresConfirmation: false,

// summarize() lists every plugin, theme and mu-plugin scanned plus the
// findings. That payload routinely exceeds maxToolResultLength, and a
// truncated security scan is worse than none.
preferSummarize: true,
} );
}

Expand Down
4 changes: 2 additions & 2 deletions src/extensions/abilities/plugin-list.js
Original file line number Diff line number Diff line change
Expand Up @@ -98,12 +98,12 @@ export function registerPluginList() {
} are inactive.\n\n`;

if ( activePlugins.length > 0 ) {
summary += ` * * Active plugins: * * ${ activePlugins.join(
summary += `**Active plugins:** ${ activePlugins.join(
', '
) }\n\n`;
}
if ( inactivePlugins.length > 0 ) {
summary += ` * * Inactive plugins: * * ${ inactivePlugins.join(
summary += `**Inactive plugins:** ${ inactivePlugins.join(
', '
) }`;
}
Expand Down
5 changes: 5 additions & 0 deletions src/extensions/abilities/security-scan.js
Original file line number Diff line number Diff line change
Expand Up @@ -95,6 +95,11 @@ export function registerSecurityScan() {
},

requiresConfirmation: false,

// summarize() renders the full pass/fail report grouped by severity.
// Re-rendering it through the LLM risks dropping findings when the
// payload exceeds maxToolResultLength.
preferSummarize: true,
} );
}

Expand Down
49 changes: 25 additions & 24 deletions src/extensions/abilities/site-health.js
Original file line number Diff line number Diff line change
Expand Up @@ -85,76 +85,72 @@ export function registerSiteHealth() {
// This creates a more natural conversational experience.

if ( msg.includes( 'php' ) ) {
return `Your PHP version is * * ${
return `Your PHP version is **${
result.php_version || 'Unknown'
} * * .`;
}**.`;
}

if ( msg.includes( 'wordpress' ) || msg.includes( 'wp version' ) ) {
return `You're running * * WordPress ${
return `You're running **WordPress ${
result.wordpress_version || 'Unknown'
} * * .`;
}**.`;
}

if (
msg.includes( 'mysql' ) ||
msg.includes( 'database' ) ||
msg.includes( 'db version' )
) {
return `Your database is * * MySQL ${
return `Your database is **MySQL ${
result.mysql_version || 'Unknown'
} * * .`;
}**.`;
}

if (
msg.includes( 'server' ) ||
msg.includes( 'nginx' ) ||
msg.includes( 'apache' )
) {
return `Your server is * * ${
return `Your server is **${
result.server_software || 'Unknown'
} * * .`;
}**.`;
}

if ( msg.includes( 'theme' ) ) {
// Handle nested object safely with optional chaining.
return `Your active theme is * * ${
return `Your active theme is **${
result.active_theme?.name || 'Unknown'
} * * (version ${ result.active_theme?.version || '?' }).`;
}** (version ${ result.active_theme?.version || '?' }).`;
}

if ( msg.includes( 'memory' ) ) {
return `Your PHP memory limit is * * ${
return `Your PHP memory limit is **${
result.memory_limit || 'Unknown'
} * * .`;
}**.`;
}

if (
msg.includes( 'url' ) ||
msg.includes( 'site address' ) ||
msg.includes( 'home' )
) {
return `Your site URL is * * ${
return `Your site URL is **${
result.site_url || result.home_url || 'Unknown'
} * * .`;
}**.`;
}

// No specific question detected - return full health summary.
// This is the default when user asks something general like "site health".
return (
`Here's your site health information:\n\n` +
` * * WordPress: * * ${
result.wordpress_version || 'Unknown'
}\n` +
` * * PHP: * * ${ result.php_version || 'Unknown' }\n` +
` * * Database: * * MySQL ${
result.mysql_version || 'Unknown'
}\n` +
` * * Server: * * ${ result.server_software || 'Unknown' }\n` +
` * * Theme: * * ${ result.active_theme?.name || 'Unknown' } (${
`**WordPress:** ${ result.wordpress_version || 'Unknown' }\n` +
`**PHP:** ${ result.php_version || 'Unknown' }\n` +
`**Database:** MySQL ${ result.mysql_version || 'Unknown' }\n` +
`**Server:** ${ result.server_software || 'Unknown' }\n` +
`**Theme:** ${ result.active_theme?.name || 'Unknown' } (${
result.active_theme?.version || '?'
})\n` +
` * * Memory Limit: * * ${ result.memory_limit || 'Unknown' }`
`**Memory Limit:** ${ result.memory_limit || 'Unknown' }`
);
},

Expand Down Expand Up @@ -209,6 +205,11 @@ export function registerSiteHealth() {

// Read-only - no confirmation needed.
requiresConfirmation: false,

// summarize() already branches on the user's question (PHP version, theme,
// memory limit, ...) and returns a complete answer, so the second LLM call
// adds nothing but a truncation risk on the full health payload.
preferSummarize: true,
} );
}

Expand Down
5 changes: 5 additions & 0 deletions src/extensions/abilities/verify-core-checksums.js
Original file line number Diff line number Diff line change
Expand Up @@ -191,6 +191,11 @@ export function registerVerifyCoreChecksums() {

// Read-only - no confirmation needed.
requiresConfirmation: false,

// summarize() emits fenced diff blocks for modified core files. The LLM
// mangles fenced code and the diffs blow past maxToolResultLength, so
// render them directly.
preferSummarize: true,
} );
}

Expand Down
97 changes: 97 additions & 0 deletions src/extensions/services/__tests__/react-agent.test.js
Original file line number Diff line number Diff line change
Expand Up @@ -497,4 +497,101 @@ describe( 'ReactAgent', () => {
expect( result.toolsUsed ).toContain( 'agentic-admin/site-health' );
} );
} );

describe( 'Structured output (grammar-constrained action envelope)', () => {
/**
* Returns the response_format of the Nth create() call, if any.
*
* @param {number} callIndex - Zero-based index of the create() call.
* @return {Object|undefined} The response_format passed, or undefined.
*/
function responseFormatOfCall( callIndex ) {
return mockEngine.chat.completions.create.mock.calls[
callIndex
]?.[ 0 ]?.response_format;
}

it( 'should be OFF by default, even when thinking is disabled', async () => {
// Default is off because it breaks thinking models: Qwen 3 emits
// `{` then whitespace to max_tokens when the grammar blocks its
// <think> block. Verified in-browser 2026-08-01, reproduced 2/2.
reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, {
disableThinking: true,
} );
reactAgent.setCallbacks( mockCallbacks );

mockStreamOnce(
mockEngine,
'{"action": "final_answer", "content": "Done"}'
);

await reactAgent.execute( 'flush the cache', [] );

expect( responseFormatOfCall( 0 ) ).toBeUndefined();
} );

it( 'should NOT constrain a thinking turn even when opted in', async () => {
// A JSON grammar cannot represent the leading <think> block.
reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, {
structuredOutput: true,
} );
reactAgent.setCallbacks( mockCallbacks );

mockStreamOnce(
mockEngine,
'{"action": "final_answer", "content": "Done"}'
);

await reactAgent.execute( 'hello', [] );

expect( responseFormatOfCall( 0 ) ).toBeUndefined();
} );

it( 'should constrain output when opted in AND thinking is disabled', async () => {
reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, {
structuredOutput: true,
disableThinking: true,
} );
reactAgent.setCallbacks( mockCallbacks );

mockStreamOnce(
mockEngine,
'{"action": "final_answer", "content": "Done"}'
);

await reactAgent.execute( 'flush the cache', [] );

const format = responseFormatOfCall( 0 );
expect( format ).toBeDefined();
expect( format.type ).toBe( 'json_object' );

// The schema is passed stringified, as WebLLM expects.
const schema = JSON.parse( format.schema );
expect( schema.required ).toEqual( [ 'action' ] );
expect( schema.properties.action.enum ).toEqual( [
'tool_call',
'final_answer',
] );
} );

it( 'should constrain follow-up turns once disableThinkingAfterTool fires', async () => {
reactAgent = new ReactAgent( mockModelLoader, mockToolRegistry, {
structuredOutput: true,
disableThinkingAfterTool: true,
} );
reactAgent.setCallbacks( mockCallbacks );

mockStreamOnce(
mockEngine,
'{"action": "tool_call", "tool": "agentic-admin/plugin-list", "args": {}}',
'{"action": "final_answer", "content": "You have 1 plugin."}'
);

await reactAgent.execute( 'list plugins', [] );

// First turn still thinks, second turn (post-tool) does not.
expect( responseFormatOfCall( 0 ) ).toBeUndefined();
expect( responseFormatOfCall( 1 ) ).toBeDefined();
} );
} );
} );
8 changes: 8 additions & 0 deletions src/extensions/services/external-engine.js
Original file line number Diff line number Diff line change
Expand Up @@ -153,6 +153,14 @@ class ExternalEngine {
model: this.modelId,
};

// WebLLM accepts a `schema` key inside response_format to grammar-constrain
// decoding. That key is WebLLM-specific: OpenAI rejects unrecognized keys
// in response_format outright, and other providers ignore it. Keep plain
// JSON mode, which every OpenAI-compatible provider understands.
if ( body.response_format?.schema ) {
body.response_format = { type: body.response_format.type };
}

// o1/o3/gpt-5 models require specific parameter mapping and don't support many standard params
if ( isModernModel ) {
if ( body.max_tokens ) {
Expand Down
9 changes: 0 additions & 9 deletions src/extensions/services/model-loader.js
Original file line number Diff line number Diff line change
Expand Up @@ -69,14 +69,6 @@ const log = createLogger( 'ModelLoader' );
*/
const DEFAULT_MODEL = 'Qwen3-1.7B-q4f16_1-MLC';

/**
* Model configuration options
*/
const MODEL_CONFIG = {
// Context window size (larger for 7B models)
context_window_size: 8192,
};

/**
* Mapping from f16 model IDs to their f32 equivalents.
* Used when the GPU does not support the shader-f16 WebGPU feature.
Expand Down Expand Up @@ -1376,7 +1368,6 @@ export {
ModelLoader,
modelLoader,
DEFAULT_MODEL,
MODEL_CONFIG,
MODEL_CONTEXT_SIZES,
ExternalEngine,
};
Expand Down
Loading
Loading