From 4c054397a0bd4fa06fccea5935a1df29c0a30895 Mon Sep 17 00:00:00 2001 From: Sarthak Agrawal Date: Mon, 5 Oct 2026 23:03:58 +0530 Subject: [PATCH 1/3] Compact managed classifier output within existing ranking budget --- README.md | 2 +- tests/compact-classifier.test.mjs | 65 +++++++++++++++++++++++++++++++ tests/production-picker.test.mjs | 2 +- worker/src/gateway-classifier.mjs | 29 ++++++++++++-- 4 files changed, 92 insertions(+), 6 deletions(-) create mode 100644 tests/compact-classifier.test.mjs diff --git a/README.md b/README.md index c4da318..dc8dbed 100644 --- a/README.md +++ b/README.md @@ -45,7 +45,7 @@ An existing **chat-completions-compatible** server can instead be configured wit Compatibility varies by provider. Set `MEME_JSON_MODE=false` only if your compatible endpoint rejects `response_format`; local validation still applies. A cloud endpoint requires HTTPS, the appropriate authorized API key, and `MEME_ALLOW_REMOTE=true`. The UI identifies a non-loopback endpoint. A loopback runtime can still invoke cloud models: use a genuinely local model for private conversations. -The managed Worker classifier now goes through the private Free AI `FleetGateway` service binding. It requests JSON with model `auto`, passes the existing fit and perspective instructions, and validates exact labels and numeric scores locally before ranking. The serious-input gate abstains when its classifier response is unavailable or malformed; regular ranking keeps the established deterministic/retrieval fallbacks. The gateway migration preserves request/response shapes and fit labels, but a general model's probability calibration is not equivalent to Jev's ordinal scoring. The source integration for Free AI Issue [#83](https://github.com/sass-maker/free-ai/issues/83) is not yet merged or deployed, so this describes the pending source path rather than a live-release change. +The production Worker classifier goes through the private Free AI `FleetGateway` service binding. It requests JSON with model `auto`, passes the existing fit and perspective instructions, and validates exact labels and numeric scores locally before ranking. Managed answers use compact `[label_index,score_0,score_1,...]` tuples, with scores requested to three decimal places, to reduce completion overhead within the existing 2,000-token limit; the adapter accepts earlier object-shaped answers and returns the same normalized classifier response. Invalid JSON logs only an allowlisted finish reason and bounded completion-token count, never prompts or provider content. Its existing single retry stays within the original ranking deadline. The serious-input gate abstains when its classifier response is unavailable or malformed; regular ranking keeps the established deterministic/retrieval fallbacks. A general model's probability calibration is not equivalent to Jev's ordinal scoring. Free AI's automatic access-denial recovery is tracked in [#95](https://github.com/sass-maker/free-ai/issues/95); end-to-end picker qualification remains in [#30](https://github.com/Significant-Hobbies/meme-lab/issues/30). The corrected 3,000-record index stores separate `meaning` and `example` vectors, for 6,000 vectors total. It contains 1,805 static meme templates and 1,195 usage-backed reaction GIFs; it contains no National Gallery artwork or empty editing canvases. The 41-item canonical coverage list includes owner-requested staples such as “My Name Is Jeff” and “I Love You 3000.” To reseed the index, confirm string metadata indexing for `view` plus boolean metadata indexing for `control` and `core`, wait for all mutations to finish, then call the tool's `/seed` endpoint in six bounded 500-record ranges (`start=0,500,…,2500&limit=500`). The `core` flag reserves ten shortlist positions for the retained baseline without preventing the broader catalogue from contributing the other twenty. Verify the corrected index and evaluation before switching production traffic. diff --git a/tests/compact-classifier.test.mjs b/tests/compact-classifier.test.mjs new file mode 100644 index 0000000..b793932 --- /dev/null +++ b/tests/compact-classifier.test.mjs @@ -0,0 +1,65 @@ +import test from 'node:test'; +import assert from 'node:assert/strict'; +import {createGatewayClassifierFetch} from '../worker/src/gateway-classifier.mjs'; + +const init=(count=30)=>({body:JSON.stringify({inputs:Array(count).fill('private input'),labels:['wrong','weak','plausible','strong','exact']})}); +const completion=(results,extra={})=>Response.json({choices:[{message:{content:JSON.stringify({results})},finish_reason:'stop'}],...extra}); + +test('compact full-batch answers preserve all ordinal probabilities and input order',async()=>{ + let calls=0; + const tuples=Array.from({length:30},(_,i)=>i%2?[3,0,0,.1,.8,.1]:[4,0,0,0,.2,.8]); + const adapter=createGatewayClassifierFetch({fetch:async request=>{ + calls++; + const body=await request.json(); + assert.equal(body.model,'auto');assert.equal(body.max_tokens,2000); + assert.deepEqual(body.response_format,{type:'json_object'}); + return completion(tuples); + }}); + const body=await (await adapter('',init())).json(); + assert.equal(calls,1);assert.equal(body.results.length,30); + for(let i=0;i<30;i++){ + assert.equal(body.results[i].label,i%2?'strong':'exact'); + assert.deepEqual(Object.values(body.results[i].scores),tuples[i].slice(1)); + } +}); + +test('compact perspective answers retain all 30 label scores',async()=>{ + const labels=Array.from({length:30},(_,i)=>`candidate ${i}`); + const adapter=createGatewayClassifierFetch({fetch:async()=>completion([[29,...labels.map((_,i)=>i===29?1:0)]])}); + const body=await (await adapter('',{body:JSON.stringify({inputs:['comment'],labels})})).json(); + assert.equal(body.results[0].label,'candidate 29'); + assert.equal(Object.keys(body.results[0].scores).length,30); + assert.equal(body.results[0].scores['candidate 29'],1); +}); + +test('invalid compact tuples never become fit scores and retries remain bounded',async()=>{ + for(const tuple of [[4,1],[5,0,0,0,0,1],[4,0,0,0,-.1,1.1],[4,0,0,0,0,0],[4,0,0,0,0,'1']]){ + let calls=0; + const adapter=createGatewayClassifierFetch({fetch:async()=>{calls++;return completion([tuple]);}}); + await assert.rejects(adapter('',init(1)),/category scores/); + assert.equal(calls,2); + } +}); + +test('truncated JSON recovers once and logs termination metadata without content',async()=>{ + const warnings=[];const original=console.warn;console.warn=message=>warnings.push(JSON.parse(message)); + let calls=0; + try { + const adapter=createGatewayClassifierFetch({fetch:async()=>++calls===1? + Response.json({choices:[{message:{content:'{"results":[ private prompt and secret'},finish_reason:'length'}],usage:{completion_tokens:2000},private:'private provider body'}):completion([[4,0,0,0,0,1]])}); + assert.equal((await adapter('',init(1))).status,200);assert.equal(calls,2); + assert.deepEqual(warnings.map(({event,failure_kind,finish_reason,completion_tokens})=>({event,failure_kind,finish_reason,completion_tokens})),[{event:'classifier_output_invalid',failure_kind:'invalid_json',finish_reason:'length',completion_tokens:2000}]); + assert.doesNotMatch(JSON.stringify(warnings),/private|prompt|secret|provider body|results/); + }finally{console.warn=original;} +}); + +test('unknown provider finish reasons and token metadata are excluded',async()=>{ + const warnings=[];const original=console.warn;console.warn=message=>warnings.push(JSON.parse(message)); + try { + const adapter=createGatewayClassifierFetch({fetch:async()=>Response.json({choices:[{message:{content:'invalid'},finish_reason:'private account detail'}],usage:{completion_tokens:'private key'}})}); + await assert.rejects(adapter('',init(1)),/invalid structured output/); + assert.equal(warnings.length,2); + for(const warning of warnings){assert.equal(warning.finish_reason,null);assert.equal(warning.completion_tokens,null);} + assert.doesNotMatch(JSON.stringify(warnings),/private|account|key/); + }finally{console.warn=original;} +}); diff --git a/tests/production-picker.test.mjs b/tests/production-picker.test.mjs index 671bb54..cc45aca 100644 --- a/tests/production-picker.test.mjs +++ b/tests/production-picker.test.mjs @@ -63,7 +63,7 @@ test('real perspective ranking accepts 30 candidates through the managed adapter const labels=JSON.parse(prompt.split('Labels by index: ')[1].split('\nInputs: ')[0]); const inputs=JSON.parse(prompt.split('\nInputs: ')[1]); requests.push({labels:labels.length,inputs:inputs.length}); - return completion(inputs.map((_,index)=>({label_index:index===0?labels.length-1:Math.max(1,labels.length-2),scores:labels.map((_,labelIndex)=>labelIndex/(labels.length*2)+index/100)}))); + return completion(inputs.map((_,index)=>[index===0?labels.length-1:Math.max(1,labels.length-2),...labels.map((_,labelIndex)=>labelIndex/(labels.length*2)+index/100)])); }}); const ranked=await rankCandidatesByPerspective('I told my coworker the deadline was today and he started another coffee break.',catalogue.slice(0,30),{fetchImpl:adapter,limit:5}); assert.equal(ranked.length,5); diff --git a/worker/src/gateway-classifier.mjs b/worker/src/gateway-classifier.mjs index 373d445..f166874 100644 --- a/worker/src/gateway-classifier.mjs +++ b/worker/src/gateway-classifier.mjs @@ -8,8 +8,10 @@ export function createGatewayClassifierFetch(binding,projectId='meme-lab') { // Ordinal ranking has five labels; perspective selection has one per meme. if(!inputs.length||inputs.length>30||!labels.length||labels.length>30) throw new Error('Classifier input is outside the supported bounds.'); const labelIndexes=labels.map((_,index)=>index); - const schema={type:'object',additionalProperties:false,required:['results'],properties:{results:{type:'array',minItems:inputs.length,maxItems:inputs.length,items:{type:'object',additionalProperties:false,required:['label_index','scores'],properties:{label_index:{type:'integer',enum:labelIndexes},scores:{type:'array',minItems:labels.length,maxItems:labels.length,items:{type:'number',minimum:0,maximum:1}}}}}}}; - const prompt=`Classify each input independently. Return label_index using only the listed label indexes. Return scores in the same order as the labels; each score must be between 0 and 1 and scores must form a probability distribution summing to 1, not all be equal. label_index must identify the highest-probability label. Preserve input order. Follow this output shape: ${JSON.stringify(schema)}\nInstructions: ${String(source.instructions||'').slice(0,3000)}\nLabels by index: ${JSON.stringify(labels.map((label,index)=>({index,label})))}\nInputs: ${JSON.stringify(inputs)}`; + // Tuples avoid repeating field names for every candidate in the bounded + // completion. Decode them back into the unchanged classifier contract. + const schema={type:'object',additionalProperties:false,required:['results'],properties:{results:{type:'array',minItems:inputs.length,maxItems:inputs.length,items:{type:'array',minItems:labels.length+1,maxItems:labels.length+1,prefixItems:[{type:'integer',enum:labelIndexes}],items:{type:'number',minimum:0,maximum:1}}}}}; + const prompt=`Classify each input independently. Return compact JSON without indentation or commentary. Each results entry is one flat array [label_index,score_0,score_1,...], with exactly ${labels.length+1} numbers. Return label_index using only the listed label indexes. Return scores in the same order as the labels; each score must be between 0 and 1 and scores must form a probability distribution summing to 1, not all be equal. Use at most three decimal places per score. label_index must identify the highest-probability label. Preserve input order. Follow this output shape: ${JSON.stringify(schema)}\nInstructions: ${String(source.instructions||'').slice(0,3000)}\nLabels by index: ${JSON.stringify(labels.map((label,index)=>({index,label})))}\nInputs: ${JSON.stringify(inputs)}`; const body=JSON.stringify({model:'auto',stream:false,response_format:{type:'json_object'},messages:[{role:'user',content:prompt}],max_tokens:2000}); for(let attempt=0;attempt<2;attempt++) { init.signal?.throwIfAborted(); @@ -39,11 +41,20 @@ export function createGatewayClassifierFetch(binding,projectId='meme-lab') { async function decodeClassifierResponse(response,inputs,labels,labelIndexes) { let raw; try { raw=await response.json(); } catch { throw new Error('Free AI gateway returned invalid JSON.'); } - if(typeof raw?.choices?.[0]?.message?.content!=='string') throw new Error('Free AI classifier returned invalid structured output.'); + if(typeof raw?.choices?.[0]?.message?.content!=='string') { + logInvalidOutput(raw,'missing_content'); + throw new Error('Free AI classifier returned invalid structured output.'); + } let parsed; - try { parsed=JSON.parse((raw?.choices?.[0]?.message?.content??'').replace(/^```(?:json)?\s*|\s*```$/g,'')); } catch { throw new Error('Free AI classifier returned invalid structured output.'); } + try { parsed=JSON.parse(raw.choices[0].message.content.replace(/^```(?:json)?\s*|\s*```$/g,'')); } catch { + logInvalidOutput(raw,'invalid_json'); + throw new Error('Free AI classifier returned invalid structured output.'); + } if(!Array.isArray(parsed?.results)||parsed.results.length!==inputs.length) throw new Error('Free AI classifier returned an unexpected result count.'); const results=parsed.results.map(result=>{ + // Accept previous object-shaped answers too; validate either representation + // before exposing any scores to ranking. + if(Array.isArray(result)) result={label_index:result[0],scores:result.slice(1)}; if(!Number.isInteger(result?.label_index)||!labelIndexes.includes(result.label_index)||!Array.isArray(result.scores)||result.scores.length!==labels.length||result.scores.some(score=>typeof score!=='number'||!Number.isFinite(score)||score<0||score>1)) throw new Error('Free AI classifier returned invalid category scores.'); const total=result.scores.reduce((sum,score)=>sum+score,0); if(total<=0) throw new Error('Free AI classifier returned empty category scores.'); @@ -51,3 +62,13 @@ async function decodeClassifierResponse(response,inputs,labels,labelIndexes) { }); return Response.json({results},{status:200}); } + +function logInvalidOutput(raw,failure_kind) { + const reason=raw?.choices?.[0]?.finish_reason; + const tokens=raw?.usage?.completion_tokens; + console.warn(JSON.stringify({ + event:'classifier_output_invalid',failure_kind, + finish_reason:['stop','length','content_filter','tool_calls'].includes(reason)?reason:null, + completion_tokens:Number.isInteger(tokens)&&tokens>=0&&tokens<=2000?tokens:null + })); +} From b039ee82f88dfe5a54d895bc0e2b8f4ffff40a42 Mon Sep 17 00:00:00 2001 From: Sarthak Agrawal Date: Mon, 5 Oct 2026 23:17:41 +0530 Subject: [PATCH 2/3] Derive ordinal fit labels from validated probabilities --- README.md | 2 +- tests/compact-classifier.test.mjs | 23 +++++++++++++++++++++-- tests/production-picker.test.mjs | 18 ++++++++++++++++++ tests/worker.test.mjs | 2 +- worker/src/gateway-classifier.mjs | 8 +++++++- 5 files changed, 48 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index dc8dbed..715d6fb 100644 --- a/README.md +++ b/README.md @@ -45,7 +45,7 @@ An existing **chat-completions-compatible** server can instead be configured wit Compatibility varies by provider. Set `MEME_JSON_MODE=false` only if your compatible endpoint rejects `response_format`; local validation still applies. A cloud endpoint requires HTTPS, the appropriate authorized API key, and `MEME_ALLOW_REMOTE=true`. The UI identifies a non-loopback endpoint. A loopback runtime can still invoke cloud models: use a genuinely local model for private conversations. -The production Worker classifier goes through the private Free AI `FleetGateway` service binding. It requests JSON with model `auto`, passes the existing fit and perspective instructions, and validates exact labels and numeric scores locally before ranking. Managed answers use compact `[label_index,score_0,score_1,...]` tuples, with scores requested to three decimal places, to reduce completion overhead within the existing 2,000-token limit; the adapter accepts earlier object-shaped answers and returns the same normalized classifier response. Invalid JSON logs only an allowlisted finish reason and bounded completion-token count, never prompts or provider content. Its existing single retry stays within the original ranking deadline. The serious-input gate abstains when its classifier response is unavailable or malformed; regular ranking keeps the established deterministic/retrieval fallbacks. A general model's probability calibration is not equivalent to Jev's ordinal scoring. Free AI's automatic access-denial recovery is tracked in [#95](https://github.com/sass-maker/free-ai/issues/95); end-to-end picker qualification remains in [#30](https://github.com/Significant-Hobbies/meme-lab/issues/30). +The production Worker classifier goes through the private Free AI `FleetGateway` service binding. It requests JSON with model `auto`, passes the existing fit and perspective instructions, and validates exact labels and numeric scores locally before ranking. Managed answers use compact `[label_index,score_0,score_1,...]` tuples, with scores requested to three decimal places, to reduce completion overhead within the existing 2,000-token limit; the adapter accepts earlier object-shaped answers and returns the same normalized classifier response. The ordinal fit label is derived from the maximum validated probability, with lower-fit tie breaking, so a contradictory generated label index cannot misstate fit confidence; the separate safety decision is preserved. Invalid JSON logs only an allowlisted finish reason and bounded completion-token count, never prompts or provider content. Its existing single retry stays within the original ranking deadline. The serious-input gate abstains when its classifier response is unavailable or malformed; regular ranking keeps the established deterministic/retrieval fallbacks. A general model's probability calibration is not equivalent to Jev's ordinal scoring. Free AI's automatic access-denial recovery is tracked in [#95](https://github.com/sass-maker/free-ai/issues/95); end-to-end picker qualification remains in [#30](https://github.com/Significant-Hobbies/meme-lab/issues/30). The corrected 3,000-record index stores separate `meaning` and `example` vectors, for 6,000 vectors total. It contains 1,805 static meme templates and 1,195 usage-backed reaction GIFs; it contains no National Gallery artwork or empty editing canvases. The 41-item canonical coverage list includes owner-requested staples such as “My Name Is Jeff” and “I Love You 3000.” To reseed the index, confirm string metadata indexing for `view` plus boolean metadata indexing for `control` and `core`, wait for all mutations to finish, then call the tool's `/seed` endpoint in six bounded 500-record ranges (`start=0,500,…,2500&limit=500`). The `core` flag reserves ten shortlist positions for the retained baseline without preventing the broader catalogue from contributing the other twenty. Verify the corrected index and evaluation before switching production traffic. diff --git a/tests/compact-classifier.test.mjs b/tests/compact-classifier.test.mjs index b793932..acbaefc 100644 --- a/tests/compact-classifier.test.mjs +++ b/tests/compact-classifier.test.mjs @@ -2,7 +2,8 @@ import test from 'node:test'; import assert from 'node:assert/strict'; import {createGatewayClassifierFetch} from '../worker/src/gateway-classifier.mjs'; -const init=(count=30)=>({body:JSON.stringify({inputs:Array(count).fill('private input'),labels:['wrong','weak','plausible','strong','exact']})}); +const fitLabels=['0 | wrong','1 | weak','2 | plausible','3 | strong','4 | exact']; +const init=(count=30)=>({body:JSON.stringify({inputs:Array(count).fill('private input'),labels:fitLabels})}); const completion=(results,extra={})=>Response.json({choices:[{message:{content:JSON.stringify({results})},finish_reason:'stop'}],...extra}); test('compact full-batch answers preserve all ordinal probabilities and input order',async()=>{ @@ -18,7 +19,7 @@ test('compact full-batch answers preserve all ordinal probabilities and input or const body=await (await adapter('',init())).json(); assert.equal(calls,1);assert.equal(body.results.length,30); for(let i=0;i<30;i++){ - assert.equal(body.results[i].label,i%2?'strong':'exact'); + assert.equal(body.results[i].label,fitLabels[i%2?3:4]); assert.deepEqual(Object.values(body.results[i].scores),tuples[i].slice(1)); } }); @@ -32,6 +33,24 @@ test('compact perspective answers retain all 30 label scores',async()=>{ assert.equal(body.results[0].scores['candidate 29'],1); }); +test('contradictory label indexes cannot override the returned ordinal probabilities',async()=>{ + const adapter=createGatewayClassifierFetch({fetch:async()=>completion([ + [1,0,.01,.01,.43,.55], + {label_index:4,scores:[.7,.2,.1,0,0]}, + [3,0,0,.5,.5,0] + ])}); + const body=await (await adapter('',init(3))).json(); + assert.deepEqual(body.results.map(result=>result.label),[fitLabels[4],fitLabels[0],fitLabels[2]]); + assert.equal(body.results[0].scores[fitLabels[4]],.55); + assert(Math.abs(body.results[1].scores[fitLabels[0]]-.7)<1e-12); +}); + +test('ordinal reconciliation preserves the separate safety classifier decision',async()=>{ + const adapter=createGatewayClassifierFetch({fetch:async()=>completion([[1,.5,.5],[1,.6,.4]])}); + const body=await (await adapter('',{body:JSON.stringify({inputs:['ambiguous','serious'],labels:['humour','serious']})})).json(); + assert.deepEqual(body.results.map(result=>result.label),['serious','serious']); +}); + test('invalid compact tuples never become fit scores and retries remain bounded',async()=>{ for(const tuple of [[4,1],[5,0,0,0,0,1],[4,0,0,0,-.1,1.1],[4,0,0,0,0,0],[4,0,0,0,0,'1']]){ let calls=0; diff --git a/tests/production-picker.test.mjs b/tests/production-picker.test.mjs index cc45aca..813e1ab 100644 --- a/tests/production-picker.test.mjs +++ b/tests/production-picker.test.mjs @@ -85,6 +85,24 @@ const testEnv={ }; const recommend=comment=>new Request('https://example.test/api/recommend',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({comment})}); +test('public confidence follows ordinal probabilities while genuine weak results stay low',async()=>{ + for(const strong of [true,false]){ + let calls=0; + const adapter=createGatewayClassifierFetch({fetch:async request=>{ + calls++; + const {messages}=await request.json(); + const inputs=JSON.parse(messages[0].content.split('\nInputs: ')[1]); + return completion(inputs.map((_,index)=>strong&&index===0?[1,0,.01,.01,.43,.55]:[4,.7+index/1000,.2-index/1000,.1,0,0])); + }}); + const response=await worker.fetch(recommend('The meeting could have been an email.'),{...testEnv,CLASSIFIER_FETCH:adapter}); + const body=await response.json(); + assert.equal(response.status,200);assert.equal(body.degraded,false);assert.equal(calls,1); + assert.equal(body.confidence,strong?'high':'low'); + assert.equal(body.candidates[0].fit_label,strong?'exact':'weak'); + if(strong)assert.equal(body.candidates[0].score,88); + } +}); + test('upstream 502 returns an honest low-confidence semantic fallback and logs degradation',async()=>{ const warnings=[];const original=console.warn;console.warn=message=>warnings.push(JSON.parse(message)); try { diff --git a/tests/worker.test.mjs b/tests/worker.test.mjs index cf00099..a80446e 100644 --- a/tests/worker.test.mjs +++ b/tests/worker.test.mjs @@ -374,7 +374,7 @@ test('managed classifier uses attributed Free AI JSON mode and preserves validat assert.equal(gatewayBody.response_format.type,'json_object'); assert.equal(gatewayBody.stream,false); assert.equal(gatewayBody.messages[0].content.includes('keep roles'),true); - assert.equal(result.results[0].label,FIT_LABELS[2].label); + assert.equal(result.results[0].label,FIT_LABELS[4].label); assert.deepEqual(result.results[0].scores,Object.fromEntries(FIT_LABELS.map(({label},index)=>[label,index/10]))); }); diff --git a/worker/src/gateway-classifier.mjs b/worker/src/gateway-classifier.mjs index f166874..744e75e 100644 --- a/worker/src/gateway-classifier.mjs +++ b/worker/src/gateway-classifier.mjs @@ -51,6 +51,7 @@ async function decodeClassifierResponse(response,inputs,labels,labelIndexes) { throw new Error('Free AI classifier returned invalid structured output.'); } if(!Array.isArray(parsed?.results)||parsed.results.length!==inputs.length) throw new Error('Free AI classifier returned an unexpected result count.'); + const ordinalFit=labels.length===5&&labels.every((label,index)=>typeof label==='string'&&label.startsWith(`${index} | `)); const results=parsed.results.map(result=>{ // Accept previous object-shaped answers too; validate either representation // before exposing any scores to ranking. @@ -58,7 +59,12 @@ async function decodeClassifierResponse(response,inputs,labels,labelIndexes) { if(!Number.isInteger(result?.label_index)||!labelIndexes.includes(result.label_index)||!Array.isArray(result.scores)||result.scores.length!==labels.length||result.scores.some(score=>typeof score!=='number'||!Number.isFinite(score)||score<0||score>1)) throw new Error('Free AI classifier returned invalid category scores.'); const total=result.scores.reduce((sum,score)=>sum+score,0); if(total<=0) throw new Error('Free AI classifier returned empty category scores.'); - return {label:labels[result.label_index],scores:Object.fromEntries(labels.map((label,index)=>[label,result.scores[index]/total]))}; + // The redundant generated index sometimes contradicts its own scores. + // Derive ordinal fit from the validated distribution; an exact tie keeps + // the lower fit level. Preserve the separate safety gate's categorical + // decision and the generic perspective contract. + const winner=ordinalFit?result.scores.reduce((best,score,index)=>score>result.scores[best]?index:best,0):result.label_index; + return {label:labels[winner],scores:Object.fromEntries(labels.map((label,index)=>[label,result.scores[index]/total]))}; }); return Response.json({results},{status:200}); } From 3c95c1aec60c7a93267d0a5f933f022d0a8a6efd Mon Sep 17 00:00:00 2001 From: Sarthak Agrawal Date: Mon, 5 Oct 2026 23:38:57 +0530 Subject: [PATCH 3/3] Preserve valid final perspective ties and report bounded probe causes --- docs/production-picker-monitoring.md | 6 ++++++ scripts/probe-production-picker.mjs | 7 ++++-- scripts/report-production-picker.mjs | 4 ++++ tests/production-picker-reporting.test.mjs | 13 +++++++++++ tests/production-picker.test.mjs | 25 ++++++++++++++++++++++ worker/src/classification.mjs | 14 ++++++------ 6 files changed, 61 insertions(+), 8 deletions(-) diff --git a/docs/production-picker-monitoring.md b/docs/production-picker-monitoring.md index dc00a16..f073df2 100644 --- a/docs/production-picker-monitoring.md +++ b/docs/production-picker-monitoring.md @@ -20,6 +20,12 @@ source, logs or chat. The workflow fails explicitly when reporting is missing or rejected; a green required-reporting run means both the picker check and App Health ingest succeeded. +The workflow's probe receipt also records allowlisted confidence and ranking +mode, whether the API marked the response as a fallback, and whether all three +perspectives are present. Strict failure criteria are unchanged. These fields +distinguish a genuine low-confidence selection from degraded routing without +including comments, provider bodies or request headers. + App Health's Overview feed must include production error logs and warning-level `*.degraded` logs to surface these outcomes independently of its 20-request aggregate-health threshold. Overview alerts remain retained occurrences rather diff --git a/scripts/probe-production-picker.mjs b/scripts/probe-production-picker.mjs index 887d753..f1a7531 100644 --- a/scripts/probe-production-picker.mjs +++ b/scripts/probe-production-picker.mjs @@ -21,8 +21,11 @@ export async function probePicker(origin='https://memes.significanthobbies.com', &&new Set(candidates.map(candidate=>candidate.id)).size===5 &&candidates.every(candidate=>typeof candidate.id==='string'&&candidate.id&&typeof candidate.media_url==='string'&&/^https:\/\//.test(candidate.media_url)&&Number.isFinite(candidate.score)); const perspectives=new Set((candidates??[]).map(candidate=>candidate.perspective)); - const degraded=body.degraded===true||body.confidence==='low'||(fixture.id==='perspectives'&&!['self','other','situation'].every(perspective=>perspectives.has(perspective))); - results.push({case:fixture.id,status:valid&&!degraded?'passed':'failed',status_code:response.status,degraded,duration_ms:Date.now()-started}); + const perspectives_complete=['self','other','situation'].every(perspective=>perspectives.has(perspective)); + const degraded=body.degraded===true||body.confidence==='low'||(fixture.id==='perspectives'&&!perspectives_complete); + const confidence=['high','medium','low'].includes(body.confidence)?body.confidence:null; + const ranking_mode=['general','perspective','general_fallback','retrieval_fallback'].includes(body.ranking_mode)?body.ranking_mode:null; + results.push({case:fixture.id,status:valid&&!degraded?'passed':'failed',status_code:response.status,degraded,fallback:body.degraded===true,confidence,ranking_mode,...(fixture.id==='perspectives'?{perspectives_complete}:{}),duration_ms:Date.now()-started}); } catch { results.push({case:fixture.id,status:'failed',reason:'request_or_response_failed',duration_ms:Date.now()-started}); } diff --git a/scripts/report-production-picker.mjs b/scripts/report-production-picker.mjs index ac39386..475d304 100644 --- a/scripts/report-production-picker.mjs +++ b/scripts/report-production-picker.mjs @@ -12,6 +12,10 @@ export async function reportPickerProbe(result,{key,fetchImpl=fetch,runUrl,relea if(Number.isInteger(fixture.status_code)&&fixture.status_code>=100&&fixture.status_code<=599) props[`${fixture.case}_status_code`]=fixture.status_code; if(typeof fixture.degraded==='boolean') props[`${fixture.case}_degraded`]=fixture.degraded; + if(typeof fixture.fallback==='boolean') props[`${fixture.case}_fallback`]=fixture.fallback; + if(['high','medium','low'].includes(fixture.confidence)) props[`${fixture.case}_confidence`]=fixture.confidence; + if(['general','perspective','general_fallback','retrieval_fallback'].includes(fixture.ranking_mode)) props[`${fixture.case}_ranking_mode`]=fixture.ranking_mode; + if(fixture.case==='perspectives'&&typeof fixture.perspectives_complete==='boolean') props.perspectives_complete=fixture.perspectives_complete; if(Number.isFinite(fixture.duration_ms)&&fixture.duration_ms>=0) props[`${fixture.case}_duration_ms`]=Math.min(Math.round(fixture.duration_ms),600000); } diff --git a/tests/production-picker-reporting.test.mjs b/tests/production-picker-reporting.test.mjs index 4168109..8693733 100644 --- a/tests/production-picker-reporting.test.mjs +++ b/tests/production-picker-reporting.test.mjs @@ -39,6 +39,19 @@ test('HTTP 200 degradation still produces an error-level failed probe event',asy assert.equal(body.logs[0].props.perspectives_degraded,true); }); +test('bounded cause fields reach App Health while unknown text is excluded',async()=>{ + let body; + await reportPickerProbe({status:'failed',results:[ + {case:'perspectives',status:'failed',degraded:true,fallback:false,confidence:'low',ranking_mode:'perspective',perspectives_complete:true}, + {case:'general',status:'failed',confidence:'private secret',ranking_mode:'private prompt',fallback:'private',perspectives_complete:'private'} + ]},{key:'test-key',fetchImpl:async(_url,init)=>{body=JSON.parse(init.body);return new Response(null,{status:202});}}); + const props=body.logs[0].props; + assert.equal(props.perspectives_fallback,false);assert.equal(props.perspectives_confidence,'low'); + assert.equal(props.perspectives_ranking_mode,'perspective');assert.equal(props.perspectives_complete,true); + assert.equal(Object.hasOwn(props,'general_confidence'),false);assert.equal(Object.hasOwn(props,'general_ranking_mode'),false); + assert.doesNotMatch(JSON.stringify(body),/private|test-key/); +}); + test('an unavailable public site still generates a failed probe log without an HTTP response',async()=>{ let body; await reportPickerProbe({status:'failed',results:[{case:'general',status:'failed',reason:'request_or_response_failed',duration_ms:100}]},{ diff --git a/tests/production-picker.test.mjs b/tests/production-picker.test.mjs index 813e1ab..efa124f 100644 --- a/tests/production-picker.test.mjs +++ b/tests/production-picker.test.mjs @@ -72,6 +72,20 @@ test('real perspective ranking accepts 30 candidates through the managed adapter assert.deepEqual(requests.map(request=>request.labels),[30,30,30,5]); }); +test('equally strong final perspective fits retain the five independently ranked memes',async()=>{ + let calls=0; + const fetchImpl=async(_url,init)=>{ + const {labels,inputs}=JSON.parse(init.body);calls++; + if(labels.length===5)return Response.json({results:inputs.map(()=>({label:labels[3],scores:Object.fromEntries(labels.map((label,index)=>[label,index===3?1:0]))}))}); + const winner=(calls-1)%3; + return Response.json({results:[{label:labels[winner],scores:Object.fromEntries(labels.map((label,index)=>[label,index===winner?0.9:0.01]))}]}); + }; + const ranked=await rankCandidatesByPerspective('I told my coworker the deadline was today and he started another coffee break.',catalogue.slice(0,30),{fetchImpl,limit:5}); + assert.equal(calls,4);assert.equal(ranked.length,5);assert.equal(new Set(ranked.map(x=>x.id)).size,5); + assert.deepEqual(new Set(ranked.map(x=>x.perspective)),new Set(['self','other','situation'])); + assert(ranked.every(x=>x.fit_label==='strong'&&x.classifier_score===.75)); +}); + const budget={idFromName:name=>name,get:()=>({fetch:async(url,options)=>{ const body=JSON.parse(options.body);const vector=url.endsWith('try-debit-vectorize');const amount=vector?body.dimensions:body.neurons; return Response.json({allowed:true,used:amount,remaining:(vector?45_000_000:9500)-amount,retryAfter:0,baselineVerified:true,[vector?'monthKey':'dayKey']:new Date().toISOString().slice(0,vector?7:10)}); @@ -157,6 +171,17 @@ test('functional probe rejects an HTTP 200 fallback and missing perspectives',as } }); +test('probe distinguishes low confidence from fallback and missing perspectives without private fields',async()=>{ + const candidates=Array.from({length:5},(_,index)=>({id:`candidate-${index}`,media_url:'https://example.test/image.png',score:80,perspective:['self','other','situation'][index%3]})); + const body={decision:'meme',confidence:'low',ranking_mode:'perspective',degraded:false,candidates,private:'private situation'}; + const result=await probePicker('https://example.test',async()=>Response.json(body)); + assert.equal(result.status,'failed');assert.equal(result.results[1].fallback,false); + assert.equal(result.results[1].confidence,'low');assert.equal(result.results[1].ranking_mode,'perspective'); + assert.equal(result.results[1].perspectives_complete,true);assert.doesNotMatch(JSON.stringify(result),/private situation/); + const unknown=await probePicker('https://example.test',async()=>Response.json({...body,confidence:'private secret',ranking_mode:'private account'})); + assert.equal(unknown.results[0].confidence,null);assert.equal(unknown.results[0].ranking_mode,null); +}); + test('browser reports caught picker errors without attaching comments or error text',()=>{ const app=readFileSync(new URL('../worker/public/app.js',import.meta.url),'utf8'); assert.match(app,/appHealthLog\?\.\('recommendation.failed'/); diff --git a/worker/src/classification.mjs b/worker/src/classification.mjs index 7df0411..bec74fb 100644 --- a/worker/src/classification.mjs +++ b/worker/src/classification.mjs @@ -133,7 +133,7 @@ function directOrdinalScore(answer) { return {classifier_score:score/(FIT_CRITERIA.length-1),fit_label:FIT_LABELS[fitIndex].key}; } -async function scoreDirectCandidates(comment,candidates,{apiKey,fetchImpl,timeoutMs,instructions=FIT_INSTRUCTIONS}={}) { +async function scoreDirectCandidates(comment,candidates,{apiKey,fetchImpl,timeoutMs,instructions=FIT_INSTRUCTIONS,allowTies=false}={}) { const state=directCandidateState(comment,candidates); state.task=instructions; const questions=Object.fromEntries(candidates.map((_,index)=>[`fit_${index}`,{ @@ -143,7 +143,7 @@ async function scoreDirectCandidates(comment,candidates,{apiKey,fetchImpl,timeou }])); const answers=await askJev({state,questions,apiKey,fetchImpl,timeoutMs}); const scored=candidates.map((record,index)=>({...record,...directOrdinalScore(answers[`fit_${index}`]),retrieval_rank:index+1})); - if(scored.length>1&&scored.every(record=>record.classifier_score===scored[0].classifier_score)) throw new Error('TypeSafe returned flat ordinal scores.'); + if(!allowTies&&scored.length>1&&scored.every(record=>record.classifier_score===scored[0].classifier_score)) throw new Error('TypeSafe returned flat ordinal scores.'); return scored; } @@ -162,12 +162,12 @@ function ordinalScore(result) { }; } -async function scoreOrdinalCandidates(comment,candidates,{fetchImpl,timeoutMs,instructions=FIT_INSTRUCTIONS,inputFor=candidateInput,apiKey}={}) { - if(apiKey) return scoreDirectCandidates(comment,candidates,{apiKey,fetchImpl,timeoutMs,instructions}); +async function scoreOrdinalCandidates(comment,candidates,{fetchImpl,timeoutMs,instructions=FIT_INSTRUCTIONS,inputFor=candidateInput,apiKey,allowTies=false}={}) { + if(apiKey) return scoreDirectCandidates(comment,candidates,{apiKey,fetchImpl,timeoutMs,instructions,allowTies}); const labels=FIT_LABELS.map(({label})=>label); const results=await classifyMany({inputs:candidates.map(record=>inputFor(comment,record)),labels,instructions,fetchImpl,timeoutMs}); const scored=candidates.map((record,index)=>({...record,...ordinalScore(results[index]),retrieval_rank:index+1})); - if(scored.length>1&&scored.every(record=>record.classifier_score===scored[0].classifier_score)) throw new Error('Classifier returned flat ordinal scores.'); + if(!allowTies&&scored.length>1&&scored.every(record=>record.classifier_score===scored[0].classifier_score)) throw new Error('Classifier returned flat ordinal scores.'); return scored; } @@ -290,7 +290,9 @@ export async function rankCandidatesByPerspective(comment,candidates,{fetchImpl= if(selected.lengthright.classifier_score-left.classifier_score||left.perspective.localeCompare(right.perspective)); const rescored=await scoreOrdinalCandidates(comment,selected.slice(0,limit),{ - fetchImpl,timeoutMs,apiKey, + // These distinct memes were already selected by non-flat perspective + // rankings. Equal final fit is valid; keep their viewpoints and scores. + fetchImpl,timeoutMs,apiKey,allowTies:true, instructions:`${FIT_INSTRUCTIONS} The input declares the intended viewpoint. Judge the candidate only for that viewpoint; do not silently switch to another participant or to the event itself.`, inputFor:(text,record)=>`COMMENT: ${text}\nVIEWPOINT: ${record.perspective_label}\nCANDIDATE: ${record.name}. Meaning: ${record.message} Social dynamic: ${record.relational_pattern} Example: ${record.example_context} Avoid: ${record.near_miss_context}` });