From 9fdfddfaa3d792ad0a2de4d1884df14ad999ffd5 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 01:38:23 +0900 Subject: [PATCH 01/43] fix(opencode): same-model session checkpoint and route evidence Preserve bounded termination/missing-output checkpoints and CO#1205 route telemetry across same-model retries under orchestrator/free without replaying provider bodies or masking incomplete control as success. Co-authored-by: Cursor --- .../opencode-same-model-midabort-20260919.md | 53 +++ .../contextual_orchestrator_route_evidence.py | 110 +++++ .../ci/opencode_review_session_checkpoint.py | 408 ++++++++++++++++++ scripts/ci/run_opencode_review_model_pool.sh | 57 +++ ..._contextual_orchestrator_route_evidence.py | 74 ++++ tests/test_opencode_model_pool_runner.py | 38 ++ ...test_opencode_review_session_checkpoint.py | 179 ++++++++ 7 files changed, 919 insertions(+) create mode 100644 docs/doctoring/opencode-same-model-midabort-20260919.md create mode 100644 scripts/ci/contextual_orchestrator_route_evidence.py create mode 100644 scripts/ci/opencode_review_session_checkpoint.py create mode 100644 tests/test_contextual_orchestrator_route_evidence.py create mode 100644 tests/test_opencode_review_session_checkpoint.py diff --git a/docs/doctoring/opencode-same-model-midabort-20260919.md b/docs/doctoring/opencode-same-model-midabort-20260919.md new file mode 100644 index 0000000000..1a26ef51cb --- /dev/null +++ b/docs/doctoring/opencode-same-model-midabort-20260919.md @@ -0,0 +1,53 @@ +# OpenCode same-model mid-abort and context loss (2026-09-19) + +## Scope + +Improve OpenCode stopping mid-work or misunderstanding required outputs on +**same-model retries** under the pinned pool +`contextual-orchestrator/orchestrator/free` (`opencode.jsonc`), without model +changes. Out of scope: inflight same-head dispatch dedupe / pg-erd handshake +waste (`ContextualWisdomLab/.github#2283`). + +## Reproduction cases (live Actions) + +| Run ID | Repo | Termination | Context loss | Retry/resume | Completion judgment | +| --- | --- | --- | --- | --- | --- | +| `35452307646` | `.github` PR `#2040` | `cancelled` (`cancel-superseded-opencode-review-runs`) | In-flight review discarded on superseded head; no exported session checkpoint | Replacement dispatch queued; prior partial work not reused | Job `cancelled`, not success; no formal verdict on cancelled head | +| `35401977816` | `.github` PR `#2278` | `failure` (`opencode-review` job) | Model pool cycled without host checkpoint; partial assistant export dropped between attempts | Same-model retries restarted from full prompt only | `review_status=exhausted`; fail-closed, no synthetic APPROVE | + +Pinned model for measurement: `contextual-orchestrator/orchestrator/free` +(resolved upstream ids recorded from sidecar route evidence when present). + +## Reused surfaces + +- `scripts/ci/contextual_orchestrator_route_evidence.py` — consumer for typed + `attempts[]` / `terminal_reason` envelopes from + `ContextualWisdomLab/contextual-orchestrator#1205` (closes #1016); mirrors the + allowlist already used in `noema_review_gate.py`. +- `scripts/ci/opencode_review_session_checkpoint.py` — host-managed checkpoint + ledger (digest-only partial work, missing required outputs, route telemetry). +- `scripts/ci/run_opencode_review_model_pool.sh` — injects bounded same-model + continuation appendix on retry (`OPENCODE_SESSION_CONTINUATION_BUDGET`, default + `2`). + +## Before / after (pinned model, fixture-backed) + +| Metric | Before | After (this change) | +| --- | --- | --- | +| Same-model retry carries prior termination reason | No | Yes (checkpoint) | +| Same-model retry carries missing required outputs | No | Yes | +| Route attempt telemetry on retry | Discarded | Preserved (CO#1205 allowlist) | +| Continuation budget | Unbounded prompt replay cycles only | Explicit env budget, tested | +| False success on incomplete control | Fail-closed already | Unchanged fail-closed | +| Partial provider body replayed into prompt | N/A | Forbidden (digest only) | + +Live completion-rate deltas require a controlled replay harness on org runners; +fixture tests prove the host contract; production measurement remains open. + +## Remaining + +- Executable fresh-session loop with durable SQLite ledger (`#2068`) still needs + read-only agent boundary preserved. +- Inflight dedupe cancellation waste (`#2283`) remains lead-owned. +- Production before/after completion rate on long tool-heavy reviews is not yet + measured on live `orchestrator/free` traffic. diff --git a/scripts/ci/contextual_orchestrator_route_evidence.py b/scripts/ci/contextual_orchestrator_route_evidence.py new file mode 100644 index 0000000000..8401342e1a --- /dev/null +++ b/scripts/ci/contextual_orchestrator_route_evidence.py @@ -0,0 +1,110 @@ +"""Consume bounded contextual-orchestrator route attempt telemetry. + +Typed ``attempts[]`` and ``terminal_reason`` on gateway error envelopes were +added for ``route_once`` failover in ContextualWisdomLab/contextual-orchestrator#1205 +(closes #1016). OpenCode and Noema consumers share this allowlisted parser so +same-model retries preserve provider-outcome evidence without re-emitting raw +provider bodies. +""" + +from __future__ import annotations + +import json +import re +from typing import Any + + +SAFE_MODEL_IDENTIFIER_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:/@+-]{0,199}$") +MAX_HTTP_ERROR_BODY_BYTES = 65536 + + +def safe_model_identifier(value: Any) -> str | None: + """Return a conservative model or route identifier safe for public logs.""" + if not isinstance(value, str): + return None + candidate = value.strip() + if not SAFE_MODEL_IDENTIFIER_RE.fullmatch(candidate): + return None + return candidate + + +def extract_gateway_route_telemetry(raw: str | bytes) -> dict[str, str | int]: + """Parse allowlisted route evidence from one gateway HTTP error envelope. + + Only ``error.detail`` scalar fields and bounded ``attempts[]`` entries are + retained. Malformed, oversized, or present-but-unsafe ``model`` / + ``terminal_reason`` values fail closed to an empty mapping so allowlisted + attempt fields cannot launder free-form error text. + """ + if isinstance(raw, bytes): + if len(raw) > MAX_HTTP_ERROR_BODY_BYTES: + return {} + try: + raw_text = raw.decode("utf-8") + except UnicodeDecodeError: + return {} + else: + raw_text = raw + try: + payload = json.loads(raw_text) + except (json.JSONDecodeError, TypeError, ValueError): + return {} + if not isinstance(payload, dict): + return {} + error = payload.get("error") + if not isinstance(error, dict): + return {} + detail = error.get("detail") + if not isinstance(detail, dict): + return {} + raw_model = detail.get("model") + if raw_model is not None: + model = safe_model_identifier(raw_model) + if model is None: + return {} + else: + model = None + raw_terminal_reason = detail.get("terminal_reason") + if raw_terminal_reason is not None: + terminal_reason = safe_model_identifier(raw_terminal_reason) + if terminal_reason is None: + return {} + else: + terminal_reason = None + telemetry: dict[str, str | int] = {} + attempts = detail.get("attempts") + if model is not None: + telemetry["served_model"] = model + if terminal_reason is not None: + telemetry["terminal_reason"] = terminal_reason + if isinstance(attempts, list) and attempts and len(attempts) <= 64: + telemetry["provider_attempt_count"] = len(attempts) + last_attempt = attempts[-1] + if isinstance(last_attempt, dict): + provider_name = safe_model_identifier(last_attempt.get("provider_name")) + phase = safe_model_identifier(last_attempt.get("phase")) + attempt_number = last_attempt.get("attempt_number") + provider_status = last_attempt.get("provider_status") + if provider_name is not None: + telemetry["provider_name"] = provider_name + if phase is not None: + telemetry["upstream_phase"] = phase + if type(attempt_number) is int and 1 <= attempt_number <= 64: + telemetry["attempt_number"] = attempt_number + if type(provider_status) is int and 100 <= provider_status <= 599: + telemetry["upstream_status"] = provider_status + return telemetry + + +def format_route_telemetry(telemetry: dict[str, str | int]) -> str: + """Format allowlisted route telemetry for bounded Actions logs.""" + ordered_keys = ( + "provider_attempt_count", + "provider_name", + "upstream_phase", + "attempt_number", + "upstream_status", + "terminal_reason", + "served_model", + ) + return " ".join(f"{key}={telemetry[key]}" for key in ordered_keys if key in telemetry) diff --git a/scripts/ci/opencode_review_session_checkpoint.py b/scripts/ci/opencode_review_session_checkpoint.py new file mode 100644 index 0000000000..b390cb444b --- /dev/null +++ b/scripts/ci/opencode_review_session_checkpoint.py @@ -0,0 +1,408 @@ +#!/usr/bin/env python3 +"""Host-managed OpenCode same-model session checkpoints and continuations. + +The trusted review host records partial attempt state and injects bounded +resume context on same-model retries. Checkpoints never grant approval +authority and never replay provider-controlled bodies into prompts. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import re +import sys +from collections.abc import Mapping, Sequence +from pathlib import Path +from typing import Any + +_REPO_ROOT = Path(__file__).resolve().parents[2] +if str(_REPO_ROOT) not in sys.path: + sys.path.insert(0, str(_REPO_ROOT)) + +from scripts.ci.contextual_orchestrator_route_evidence import ( # noqa: E402 + extract_gateway_route_telemetry, + format_route_telemetry, +) + + +CHECKPOINT_SCHEMA = 1 +CONTROL_SENTINEL = "opencode-review-control-v1" +REQUIRED_OUTPUT_MARKERS = ( + CONTROL_SENTINEL, + "adversarial_validation", + '"result"', + "Developer experience:", + "User experience:", +) +TERMINATION_PATTERNS: tuple[tuple[str, re.Pattern[str]], ...] = ( + ("provider-fatal", re.compile(r"contextoverflowerror|tokens_limit_reached|model_not_found|no endpoints", re.I)), + ("provider-timeout", re.compile(r"timed?\ ?out|timeout", re.I)), + ("provider-rate-limit", re.compile(r"rate.?limit|too many requests|\b429\b", re.I)), + ("provider-error", re.compile(r'"type"\s*:\s*"error"', re.I)), + ("export-empty", re.compile(r"assistant-empty-export", re.I)), + ("invalid-control", re.compile(r"invalid-control-output", re.I)), + ("sessionless", re.compile(r"sessionless-json", re.I)), + ("nonzero-exit", re.compile(r"exit", re.I)), +) +MAX_PARTIAL_DIGEST_CHARS = 64 +MAX_CONTINUATION_BYTES = 8192 +DEFAULT_CONTINUATION_BUDGET = 2 + + +def _read_bounded_text(path: Path, max_bytes: int) -> str: + """Read at most ``max_bytes`` from a file as UTF-8 replacement text.""" + if not path.is_file(): + return "" + data = path.read_bytes()[:max_bytes] + return data.decode("utf-8", errors="replace") + + +def classify_termination( + *, + json_path: Path, + export_path: Path, + exit_code: int, + stderr_path: Path | None = None, + log_hint: str = "", +) -> str: + """Return a bounded termination reason for one OpenCode attempt.""" + combined = "\n".join( + part + for part in ( + _read_bounded_text(json_path, 65536), + _read_bounded_text(export_path, 65536), + _read_bounded_text(stderr_path, 65536) if stderr_path else "", + log_hint, + ) + if part + ) + for label, pattern in TERMINATION_PATTERNS: + if pattern.search(combined): + return label + if exit_code != 0: + return "nonzero-exit" + if not summarize_partial_assistant(export_path).get("assistant_text_present"): + return "export-empty" + return "incomplete-control" + + +def summarize_partial_assistant(export_path: Path) -> dict[str, str | int | bool]: + """Return bounded metadata about partial assistant output, never raw text.""" + if not export_path.is_file(): + return { + "assistant_text_present": False, + "assistant_line_count": 0, + "assistant_sha256": "", + "has_control_sentinel": False, + } + try: + payload = json.loads(export_path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError, UnicodeDecodeError): + return { + "assistant_text_present": False, + "assistant_line_count": 0, + "assistant_sha256": "", + "has_control_sentinel": False, + } + texts: list[str] = [] + if isinstance(payload, dict): + messages = payload.get("messages") + if isinstance(messages, list): + for message in messages: + if not isinstance(message, dict): + continue + info = message.get("info") + if not isinstance(info, dict) or info.get("role") != "assistant": + continue + parts = message.get("parts") + if not isinstance(parts, list): + continue + for part in parts: + if isinstance(part, dict) and part.get("type") == "text": + text = part.get("text") + if isinstance(text, str) and text.strip(): + texts.append(text) + joined = "\n".join(texts) + digest = hashlib.sha256(joined.encode("utf-8")).hexdigest() if joined else "" + return { + "assistant_text_present": bool(joined.strip()), + "assistant_line_count": len(joined.splitlines()) if joined else 0, + "assistant_sha256": digest[:MAX_PARTIAL_DIGEST_CHARS], + "has_control_sentinel": CONTROL_SENTINEL in joined, + } + + +def missing_required_outputs(partial_text: str) -> list[str]: + """Return required review output markers absent from partial assistant text.""" + missing: list[str] = [] + for marker in REQUIRED_OUTPUT_MARKERS: + if marker not in partial_text: + missing.append(marker) + return missing + + +def _load_checkpoint(path: Path) -> dict[str, Any]: + """Load an existing checkpoint or return an empty document.""" + if not path.is_file(): + return {"schema": CHECKPOINT_SCHEMA, "attempts": []} + try: + loaded = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError, UnicodeDecodeError): + return {"schema": CHECKPOINT_SCHEMA, "attempts": []} + if not isinstance(loaded, dict): + return {"schema": CHECKPOINT_SCHEMA, "attempts": []} + attempts = loaded.get("attempts") + if not isinstance(attempts, list): + loaded["attempts"] = [] + loaded.setdefault("schema", CHECKPOINT_SCHEMA) + return loaded + + +def record_attempt_checkpoint( + *, + checkpoint_path: Path, + model_candidate: str, + attempt: int, + head_sha: str, + run_id: str, + run_attempt: str, + json_path: Path, + export_path: Path, + exit_code: int, + stderr_path: Path | None = None, + route_evidence_path: Path | None = None, + log_hint: str = "", +) -> dict[str, Any]: + """Append one bounded attempt record to the host checkpoint ledger.""" + partial_summary = summarize_partial_assistant(export_path) + partial_text = "" + if export_path.is_file(): + try: + payload = json.loads(export_path.read_text(encoding="utf-8")) + if isinstance(payload, dict): + messages = payload.get("messages") + if isinstance(messages, list): + chunks: list[str] = [] + for message in messages: + if not isinstance(message, dict): + continue + parts = message.get("parts") + if not isinstance(parts, list): + continue + for part in parts: + if isinstance(part, dict) and isinstance(part.get("text"), str): + chunks.append(part["text"]) + partial_text = "\n".join(chunks) + except (OSError, json.JSONDecodeError, UnicodeDecodeError): + partial_text = "" + route_telemetry: dict[str, str | int] = {} + if route_evidence_path and route_evidence_path.is_file(): + route_telemetry = extract_gateway_route_telemetry( + route_evidence_path.read_bytes()[:65536] + ) + entry = { + "attempt": attempt, + "model_candidate": model_candidate, + "head_sha": head_sha, + "run_id": run_id, + "run_attempt": run_attempt, + "termination_reason": classify_termination( + json_path=json_path, + export_path=export_path, + exit_code=exit_code, + stderr_path=stderr_path, + log_hint=log_hint, + ), + "exit_code": exit_code, + "partial_summary": partial_summary, + "missing_required_outputs": missing_required_outputs(partial_text), + "route_telemetry": route_telemetry, + } + document = _load_checkpoint(checkpoint_path) + attempts = document.setdefault("attempts", []) + if isinstance(attempts, list): + attempts.append(entry) + document["pinned_model"] = model_candidate + document["head_sha"] = head_sha + if route_telemetry: + document["last_route_telemetry"] = route_telemetry + checkpoint_path.parent.mkdir(parents=True, exist_ok=True) + checkpoint_path.write_text( + json.dumps(document, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + return entry + + +def continuation_budget_remaining( + checkpoint_path: Path, *, budget: int = DEFAULT_CONTINUATION_BUDGET +) -> int: + """Return how many same-model continuations remain for this checkpoint.""" + document = _load_checkpoint(checkpoint_path) + attempts = document.get("attempts") + used = len(attempts) - 1 if isinstance(attempts, list) and attempts else 0 + if used < 0: + used = 0 + return max(0, budget - used) + + +def build_continuation_appendix(checkpoint_path: Path, *, budget: int) -> str: + """Build a bounded same-model continuation appendix from checkpoint evidence.""" + document = _load_checkpoint(checkpoint_path) + attempts = document.get("attempts") + if not isinstance(attempts, list) or not attempts: + return "" + remaining = continuation_budget_remaining(checkpoint_path, budget=budget) + if remaining <= 0: + return "" + last = attempts[-1] + if not isinstance(last, Mapping): + return "" + lines = [ + "", + "## Same-model continuation (host checkpoint; not approval evidence)", + f"- Pinned model: `{document.get('pinned_model', 'unknown')}`", + f"- Prior attempt: `{last.get('attempt', '?')}`", + f"- Termination reason: `{last.get('termination_reason', 'unknown')}`", + f"- Continuation budget remaining after this attempt: `{remaining}`", + ] + partial = last.get("partial_summary") + if isinstance(partial, Mapping): + lines.append( + "- Partial assistant digest: " + f"lines=`{partial.get('assistant_line_count', 0)}` " + f"sha256=`{partial.get('assistant_sha256', '')}` " + f"control_sentinel=`{partial.get('has_control_sentinel', False)}`" + ) + missing = last.get("missing_required_outputs") + if isinstance(missing, list) and missing: + lines.append( + "- Required outputs still missing from the prior attempt: " + + ", ".join(f"`{item}`" for item in missing[:8]) + ) + route = last.get("route_telemetry") + if isinstance(route, Mapping) and route: + lines.append(f"- Route evidence: {format_route_telemetry(dict(route))}") + lines.extend( + [ + "- Resume from the trusted evidence packet and complete every required output.", + "- Do not treat this appendix as approval evidence or permission to omit probes.", + "- Return exactly one final control block for the current head when complete.", + "", + ] + ) + appendix = "\n".join(lines) + encoded = appendix.encode("utf-8") + if len(encoded) <= MAX_CONTINUATION_BYTES: + return appendix + return encoded[:MAX_CONTINUATION_BYTES].decode("utf-8", errors="ignore") + + +def append_continuation_to_prompt( + prompt_path: Path, checkpoint_path: Path, *, budget: int +) -> int: + """Append the continuation appendix to ``prompt_path`` when budget allows.""" + appendix = build_continuation_appendix(checkpoint_path, budget=budget) + if not appendix: + return continuation_budget_remaining(checkpoint_path, budget=budget) + existing = prompt_path.read_text(encoding="utf-8") if prompt_path.is_file() else "" + prompt_path.write_text(existing + appendix, encoding="utf-8") + return continuation_budget_remaining(checkpoint_path, budget=budget) + + +def parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: + """Parse checkpoint CLI commands.""" + parser = argparse.ArgumentParser(description=__doc__) + subparsers = parser.add_subparsers(dest="command", required=True) + + record = subparsers.add_parser("record", help="Record one failed attempt checkpoint.") + record.add_argument("--checkpoint", required=True, type=Path) + record.add_argument("--model-candidate", required=True) + record.add_argument("--attempt", required=True, type=int) + record.add_argument("--head-sha", required=True) + record.add_argument("--run-id", required=True) + record.add_argument("--run-attempt", required=True) + record.add_argument("--json-path", required=True, type=Path) + record.add_argument("--export-path", required=True, type=Path) + record.add_argument("--exit-code", required=True, type=int) + record.add_argument("--stderr-path", type=Path) + record.add_argument("--route-evidence-path", type=Path) + record.add_argument("--log-hint", default="") + + append = subparsers.add_parser( + "append-continuation", help="Append a bounded continuation appendix to a prompt." + ) + append.add_argument("--prompt", required=True, type=Path) + append.add_argument("--checkpoint", required=True, type=Path) + append.add_argument( + "--budget", + type=int, + default=int( + __import__("os").environ.get( + "OPENCODE_SESSION_CONTINUATION_BUDGET", + str(DEFAULT_CONTINUATION_BUDGET), + ) + ), + ) + + budget = subparsers.add_parser( + "budget-remaining", help="Print remaining same-model continuation budget." + ) + budget.add_argument("--checkpoint", required=True, type=Path) + budget.add_argument( + "--budget", + type=int, + default=int( + __import__("os").environ.get( + "OPENCODE_SESSION_CONTINUATION_BUDGET", + str(DEFAULT_CONTINUATION_BUDGET), + ) + ), + ) + return parser.parse_args(argv) + + +def main(argv: Sequence[str] | None = None) -> int: + """Execute one checkpoint subcommand.""" + args = parse_args(argv) + if args.command == "record": + entry = record_attempt_checkpoint( + checkpoint_path=args.checkpoint, + model_candidate=args.model_candidate, + attempt=args.attempt, + head_sha=args.head_sha, + run_id=args.run_id, + run_attempt=args.run_attempt, + json_path=args.json_path, + export_path=args.export_path, + exit_code=args.exit_code, + stderr_path=args.stderr_path, + route_evidence_path=args.route_evidence_path, + log_hint=args.log_hint, + ) + print( + json.dumps( + { + "termination_reason": entry["termination_reason"], + "route_telemetry": format_route_telemetry(entry["route_telemetry"]), + }, + separators=(",", ":"), + ) + ) + return 0 + if args.command == "append-continuation": + remaining = append_continuation_to_prompt( + args.prompt, args.checkpoint, budget=args.budget + ) + print(remaining) + return 0 + if args.command == "budget-remaining": + print(continuation_budget_remaining(args.checkpoint, budget=args.budget)) + return 0 + raise SystemExit(f"unknown command: {args.command}") + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/ci/run_opencode_review_model_pool.sh b/scripts/ci/run_opencode_review_model_pool.sh index 80f57d1d43..45b90b3899 100644 --- a/scripts/ci/run_opencode_review_model_pool.sh +++ b/scripts/ci/run_opencode_review_model_pool.sh @@ -190,6 +190,48 @@ PY } >"$prompt_file" } +record_session_checkpoint() { + local model_candidate="$1" + local attempt="$2" + local checkpoint_file="$3" + local json_file="$4" + local export_file="$5" + local exit_code="$6" + local stderr_file="$7" + local route_evidence_file="${8:-}" + + PYTHONPATH="$GITHUB_WORKSPACE${PYTHONPATH:+:$PYTHONPATH}" python3 "$GITHUB_WORKSPACE/scripts/ci/opencode_review_session_checkpoint.py" record \ + --checkpoint "$checkpoint_file" \ + --model-candidate "$model_candidate" \ + --attempt "$attempt" \ + --head-sha "$HEAD_SHA" \ + --run-id "$RUN_ID" \ + --run-attempt "$RUN_ATTEMPT" \ + --json-path "$json_file" \ + --export-path "$export_file" \ + --exit-code "$exit_code" \ + ${stderr_file:+--stderr-path "$stderr_file"} \ + ${route_evidence_file:+--route-evidence-path "$route_evidence_file"} \ + || true +} + +append_same_model_continuation() { + local prompt_file="$1" + local checkpoint_file="$2" + local budget="${OPENCODE_SESSION_CONTINUATION_BUDGET:-2}" + local remaining + + remaining="$(python3 "$GITHUB_WORKSPACE/scripts/ci/opencode_review_session_checkpoint.py" append-continuation \ + --prompt "$prompt_file" \ + --checkpoint "$checkpoint_file" \ + --budget "$budget" 2>/dev/null || printf '0\n')" + if [ "${remaining:-0}" -le 0 ]; then + return 1 + fi + printf 'OpenCode same-model continuation appended; budget remaining after this attempt=%s.\n' "$remaining" + return 0 +} + write_schema_repair_prompt() { local model_candidate="$1" local prompt_file="$2" @@ -536,6 +578,8 @@ main() { candidate_output_file="${RUNNER_TEMP}/opencode-review-${safe_model}.md" opencode_json_file="${candidate_output_file}.jsonl" opencode_export_file="${candidate_output_file}.session.json" + checkpoint_file="${RUNNER_TEMP}/opencode-checkpoint-${safe_model}.json" + route_evidence_file="${OPENCODE_ROUTE_EVIDENCE_FILE:-}" write_prompt "$model_candidate" "$prompt_file" effective_attempts="$attempts" if is_schema_repair_candidate "$model_candidate"; then @@ -546,6 +590,14 @@ main() { write_schema_repair_prompt "$model_candidate" "$prompt_file" printf 'OpenCode %s schema-repair attempt %s/%s will re-review from trusted evidence with a non-replayable control checklist.\n' \ "$model_candidate" "$attempt" "$effective_attempts" + elif [ "$attempt" -gt 1 ] && [ -f "$checkpoint_file" ]; then + if append_same_model_continuation "$prompt_file" "$checkpoint_file"; then + printf 'OpenCode %s same-model continuation attempt %s/%s reuses host checkpoint evidence under contextual-orchestrator/orchestrator/free.\n' \ + "$model_candidate" "$attempt" "$effective_attempts" + else + printf 'OpenCode %s same-model continuation budget exhausted; retry proceeds without checkpoint appendix.\n' \ + "$model_candidate" + fi fi if [ "$max_total_attempts" -gt 0 ] && [ "$total_attempts" -ge "$max_total_attempts" ]; then printf 'OpenCode model pool reached the per-run provider attempt ceiling of %s attempts; ending the pool to bound provider spend. Set OPENCODE_POOL_MAX_TOTAL_ATTEMPTS=0 to disable.\n' "$max_total_attempts" @@ -569,6 +621,11 @@ main() { else run_status=$? fi + if [ "$run_status" -ne 0 ] && [ "$run_status" -ne 2 ]; then + record_session_checkpoint "$model_candidate" "$attempt" "$checkpoint_file" \ + "$opencode_json_file" "$opencode_export_file" "$run_status" \ + "${opencode_json_file}.stderr" "$route_evidence_file" + fi if [ "$run_status" -ne 3 ] && is_credit_exhausted_failure "$opencode_json_file" "${opencode_json_file}.stderr"; then dead_candidate_reasons[$model_candidate]="provider credits exhausted (HTTP 402 / payment required)" printf 'OpenCode %s provider credits are exhausted; marking this candidate failed for the rest of the run so retries cannot accrue further spend.\n' "$model_candidate" diff --git a/tests/test_contextual_orchestrator_route_evidence.py b/tests/test_contextual_orchestrator_route_evidence.py new file mode 100644 index 0000000000..fca8692064 --- /dev/null +++ b/tests/test_contextual_orchestrator_route_evidence.py @@ -0,0 +1,74 @@ +"""Tests for bounded contextual-orchestrator route attempt telemetry.""" + +from __future__ import annotations + +import json + +from scripts.ci.contextual_orchestrator_route_evidence import ( + extract_gateway_route_telemetry, + format_route_telemetry, +) + + +def test_extract_gateway_route_telemetry_reads_typed_attempts() -> None: + """CO#1205 typed attempts[] fields are preserved without raw provider bodies.""" + payload = { + "error": { + "detail": { + "model": "openrouter/meta-llama/llama-3.3-70b-instruct:free", + "terminal_reason": "eligible_candidates_exhausted", + "attempts": [ + { + "provider_name": "openrouter", + "phase": "connecting", + "attempt_number": 1, + "provider_status": 429, + }, + { + "provider_name": "nvidia_nim", + "phase": "streaming", + "attempt_number": 2, + "provider_status": 502, + }, + ], + } + } + } + telemetry = extract_gateway_route_telemetry(json.dumps(payload)) + assert telemetry["served_model"] == "openrouter/meta-llama/llama-3.3-70b-instruct:free" + assert telemetry["terminal_reason"] == "eligible_candidates_exhausted" + assert telemetry["provider_attempt_count"] == 2 + assert telemetry["provider_name"] == "nvidia_nim" + assert telemetry["upstream_phase"] == "streaming" + assert telemetry["attempt_number"] == 2 + assert telemetry["upstream_status"] == 502 + + +def test_extract_gateway_route_telemetry_rejects_unsafe_identifiers() -> None: + """Free-form provider error text must not become public telemetry.""" + payload = { + "error": { + "detail": { + "model": "bad\nmodel", + "terminal_reason": "x" * 300, + "attempts": [{"provider_name": "ok", "attempt_number": 999}], + } + } + } + # Present-but-unsafe model/terminal_reason fail closed for the whole + # envelope so allowlisted attempt fields cannot launder free-form text. + assert extract_gateway_route_telemetry(json.dumps(payload)) == {} + + +def test_format_route_telemetry_is_stable() -> None: + """Formatted telemetry uses a fixed key order for log correlation.""" + formatted = format_route_telemetry( + { + "terminal_reason": "pool_exhausted", + "provider_attempt_count": 3, + "served_model": "orchestrator/free", + } + ) + assert formatted == ( + "provider_attempt_count=3 terminal_reason=pool_exhausted served_model=orchestrator/free" + ) diff --git a/tests/test_opencode_model_pool_runner.py b/tests/test_opencode_model_pool_runner.py index 2965d4c55c..3a0b268785 100644 --- a/tests/test_opencode_model_pool_runner.py +++ b/tests/test_opencode_model_pool_runner.py @@ -981,3 +981,41 @@ def test_paid_provider_does_not_gain_an_implicit_schema_repair_attempt( assert "attempt 1/1" in result.stdout assert "schema-repair attempt" not in result.stdout assert "attempt 2/" not in result.stdout + + +def test_same_model_retry_appends_host_checkpoint_continuation(tmp_path: Path) -> None: + """A second same-model attempt reuses bounded checkpoint context instead of restarting blind.""" + prompt_capture = tmp_path / "prompt.md" + result = run_failed_model( + tmp_path, + json_line='{"type":"step_start","sessionID":"session-1"}', + model_candidates="contextual-orchestrator/orchestrator/free", + prompt_capture=prompt_capture, + extra_env={ + "FAKE_OPENCODE_RUN_EXIT": "0", + "FAKE_OPENCODE_EXPORT": json.dumps( + { + "messages": [ + { + "info": {"role": "assistant"}, + "parts": [ + {"type": "text", "text": "partial review without control block"} + ], + } + ] + } + ), + "OPENCODE_MODEL_ATTEMPTS": "2", + "OPENCODE_BACKOFF_INITIAL_SECONDS": "0", + "OPENCODE_POOL_MAX_CYCLES": "1", + "OPENCODE_POOL_CYCLE_SLEEP_SECONDS": "0", + "OPENCODE_SESSION_CONTINUATION_BUDGET": "2", + }, + ) + + assert result.returncode == 1 + assert "same-model continuation attempt 2/2" in result.stdout + prompt_text = prompt_capture.read_text(encoding="utf-8") + assert "Same-model continuation (host checkpoint" in prompt_text + assert "partial review without control block" not in prompt_text + assert "contextual-orchestrator/orchestrator/free" in prompt_text diff --git a/tests/test_opencode_review_session_checkpoint.py b/tests/test_opencode_review_session_checkpoint.py new file mode 100644 index 0000000000..df13947dcd --- /dev/null +++ b/tests/test_opencode_review_session_checkpoint.py @@ -0,0 +1,179 @@ +"""Tests for host-managed OpenCode same-model session checkpoints.""" + +from __future__ import annotations + +import json +import subprocess +import sys +from pathlib import Path + +from scripts.ci.opencode_review_session_checkpoint import ( + append_continuation_to_prompt, + build_continuation_appendix, + classify_termination, + continuation_budget_remaining, + missing_required_outputs, + record_attempt_checkpoint, + summarize_partial_assistant, +) + + +def _export_with_text(text: str) -> str: + return json.dumps( + { + "messages": [ + { + "info": {"role": "assistant"}, + "parts": [{"type": "text", "text": text}], + } + ] + } + ) + + +def test_summarize_partial_assistant_never_returns_raw_text(tmp_path: Path) -> None: + """Checkpoint metadata stays digest-only.""" + export_path = tmp_path / "export.json" + export_path.write_text( + _export_with_text("secret partial body\nopencode-review-control-v1"), + encoding="utf-8", + ) + summary = summarize_partial_assistant(export_path) + assert summary["assistant_text_present"] is True + assert summary["has_control_sentinel"] is True + assert "secret" not in json.dumps(summary) + + +def test_missing_required_outputs_lists_absent_markers() -> None: + """Incomplete control output records which contract markers are still missing.""" + missing = missing_required_outputs("partial progress only") + assert "opencode-review-control-v1" in missing + assert "adversarial_validation" in missing + + +def test_classify_termination_detects_provider_fatal(tmp_path: Path) -> None: + """Fatal provider signatures classify as bounded termination reasons.""" + json_path = tmp_path / "run.jsonl" + json_path.write_text( + '{"type":"error","error":{"name":"ContextOverflowError","data":{}}}\n', + encoding="utf-8", + ) + reason = classify_termination( + json_path=json_path, + export_path=tmp_path / "missing.json", + exit_code=1, + ) + assert reason == "provider-fatal" + + +def test_record_and_continue_preserves_route_evidence(tmp_path: Path) -> None: + """Same-model retries carry route telemetry without replaying provider bodies.""" + export_path = tmp_path / "export.json" + export_path.write_text(_export_with_text("in progress"), encoding="utf-8") + route_path = tmp_path / "route.json" + route_path.write_text( + json.dumps( + { + "error": { + "detail": { + "model": "orchestrator/free", + "terminal_reason": "rate_limited", + "attempts": [ + { + "provider_name": "openrouter", + "phase": "connecting", + "attempt_number": 1, + "provider_status": 429, + } + ], + } + } + } + ), + encoding="utf-8", + ) + checkpoint_path = tmp_path / "checkpoint.json" + record_attempt_checkpoint( + checkpoint_path=checkpoint_path, + model_candidate="contextual-orchestrator/orchestrator/free", + attempt=1, + head_sha="a" * 40, + run_id="35401977816", + run_attempt="1", + json_path=tmp_path / "run.jsonl", + export_path=export_path, + exit_code=1, + route_evidence_path=route_path, + ) + appendix = build_continuation_appendix(checkpoint_path, budget=2) + assert "contextual-orchestrator/orchestrator/free" in appendix + assert "termination reason" in appendix.casefold() + assert "route evidence" in appendix.casefold() + assert "provider_attempt_count=1" in appendix + assert "in progress" not in appendix + + +def test_append_continuation_respects_budget(tmp_path: Path) -> None: + """Continuation budget is explicit and fails closed when exhausted.""" + export_path = tmp_path / "export.json" + export_path.write_text(_export_with_text("partial"), encoding="utf-8") + checkpoint_path = tmp_path / "checkpoint.json" + prompt_path = tmp_path / "prompt.md" + prompt_path.write_text("base prompt\n", encoding="utf-8") + for attempt in (1, 2, 3): + record_attempt_checkpoint( + checkpoint_path=checkpoint_path, + model_candidate="contextual-orchestrator/orchestrator/free", + attempt=attempt, + head_sha="b" * 40, + run_id="35452307646", + run_attempt="1", + json_path=tmp_path / f"run-{attempt}.jsonl", + export_path=export_path, + exit_code=1, + ) + assert continuation_budget_remaining(checkpoint_path, budget=2) == 0 + before = prompt_path.read_text(encoding="utf-8") + remaining = append_continuation_to_prompt(prompt_path, checkpoint_path, budget=2) + assert remaining == 0 + assert prompt_path.read_text(encoding="utf-8") == before + + +def test_cli_record_emits_bounded_json(tmp_path: Path) -> None: + """The checkpoint CLI prints only bounded metadata to stdout.""" + export_path = tmp_path / "export.json" + export_path.write_text(_export_with_text("partial"), encoding="utf-8") + checkpoint_path = tmp_path / "checkpoint.json" + completed = subprocess.run( + [ + sys.executable, + "scripts/ci/opencode_review_session_checkpoint.py", + "record", + "--checkpoint", + str(checkpoint_path), + "--model-candidate", + "contextual-orchestrator/orchestrator/free", + "--attempt", + "1", + "--head-sha", + "c" * 40, + "--run-id", + "35452307646", + "--run-attempt", + "1", + "--json-path", + str(tmp_path / "run.jsonl"), + "--export-path", + str(export_path), + "--exit-code", + "1", + ], + cwd=Path(__file__).resolve().parents[1], + capture_output=True, + text=True, + check=False, + ) + assert completed.returncode == 0 + payload = json.loads(completed.stdout.strip()) + assert "termination_reason" in payload + assert "partial" not in completed.stdout From cfeda192f68af43077cae3b3cd61ef8f11eaea20 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 02:04:12 +0900 Subject: [PATCH 02/43] test(opencode): reproduce unsafe checkpoint evidence reads --- ...test_opencode_review_session_checkpoint.py | 65 +++++++++++++++++++ 1 file changed, 65 insertions(+) diff --git a/tests/test_opencode_review_session_checkpoint.py b/tests/test_opencode_review_session_checkpoint.py index df13947dcd..0dc3d8dc65 100644 --- a/tests/test_opencode_review_session_checkpoint.py +++ b/tests/test_opencode_review_session_checkpoint.py @@ -8,6 +8,7 @@ from pathlib import Path from scripts.ci.opencode_review_session_checkpoint import ( + _read_bounded_text, append_continuation_to_prompt, build_continuation_appendix, classify_termination, @@ -31,6 +32,70 @@ def _export_with_text(text: str) -> str: ) +def test_read_bounded_text_reads_only_the_declared_prefix( + monkeypatch, tmp_path: Path +) -> None: + """A bounded log read must not materialize the complete file first.""" + log_path = tmp_path / "large.log" + log_path.write_bytes(b"safe" + b"x" * 131_072) + + def forbid_unbounded_read(_path: Path) -> bytes: + raise AssertionError("Path.read_bytes() materialized the complete log") + + monkeypatch.setattr(Path, "read_bytes", forbid_unbounded_read) + assert _read_bounded_text(log_path, 4) == "safe" + + +def test_record_attempt_ignores_user_prompt_contract_markers(tmp_path: Path) -> None: + """Only assistant output may satisfy the continuation output contract.""" + export_path = tmp_path / "export.json" + export_path.write_text( + json.dumps( + { + "messages": [ + { + "info": {"role": "user"}, + "parts": [ + { + "type": "text", + "text": ( + "opencode-review-control-v1 adversarial_validation " + '"result" Developer experience: User experience:' + ), + } + ], + }, + { + "info": {"role": "assistant"}, + "parts": [{"type": "text", "text": "partial review"}], + }, + ] + } + ), + encoding="utf-8", + ) + checkpoint_path = tmp_path / "checkpoint.json" + entry = record_attempt_checkpoint( + checkpoint_path=checkpoint_path, + model_candidate="contextual-orchestrator/orchestrator/free", + attempt=1, + head_sha="d" * 40, + run_id="35401977816", + run_attempt="1", + json_path=tmp_path / "run.jsonl", + export_path=export_path, + exit_code=1, + ) + + assert entry["missing_required_outputs"] == [ + "opencode-review-control-v1", + "adversarial_validation", + '"result"', + "Developer experience:", + "User experience:", + ] + + def test_summarize_partial_assistant_never_returns_raw_text(tmp_path: Path) -> None: """Checkpoint metadata stays digest-only.""" export_path = tmp_path / "export.json" From afe1420d3175d81f51c091b3499d792037cc304e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 02:04:34 +0900 Subject: [PATCH 03/43] test(opencode): reproduce continuation scope and accumulation defects --- tests/test_opencode_model_pool_runner.py | 70 ++++++++++++++++++++++++ 1 file changed, 70 insertions(+) diff --git a/tests/test_opencode_model_pool_runner.py b/tests/test_opencode_model_pool_runner.py index 3a0b268785..430f75942f 100644 --- a/tests/test_opencode_model_pool_runner.py +++ b/tests/test_opencode_model_pool_runner.py @@ -983,6 +983,76 @@ def test_paid_provider_does_not_gain_an_implicit_schema_repair_attempt( assert "attempt 2/" not in result.stdout +def test_non_orchestrator_candidate_does_not_receive_checkpoint_continuation( + tmp_path: Path, +) -> None: + """Checkpoint continuation remains scoped to the pinned orchestrator/free route.""" + prompt_capture = tmp_path / "prompt.md" + result = run_failed_model( + tmp_path, + json_line='{"type":"step_start","sessionID":"session-1"}', + model_candidates="github-models/openai/gpt-5", + prompt_capture=prompt_capture, + extra_env={ + "FAKE_OPENCODE_RUN_EXIT": "0", + "FAKE_OPENCODE_EXPORT": json.dumps( + { + "messages": [ + { + "info": {"role": "assistant"}, + "parts": [{"type": "text", "text": "partial review"}], + } + ] + } + ), + "OPENCODE_MODEL_ATTEMPTS": "2", + "OPENCODE_BACKOFF_INITIAL_SECONDS": "0", + "OPENCODE_POOL_MAX_CYCLES": "1", + "OPENCODE_POOL_CYCLE_SLEEP_SECONDS": "0", + }, + ) + + assert result.returncode == 1 + assert "same-model continuation attempt" not in result.stdout + assert "Same-model continuation (host checkpoint" not in prompt_capture.read_text( + encoding="utf-8" + ) + + +def test_latest_retry_replaces_prior_checkpoint_appendix(tmp_path: Path) -> None: + """Each retry prompt carries one latest checkpoint appendix, not accumulated history.""" + prompt_capture = tmp_path / "prompt.md" + result = run_failed_model( + tmp_path, + json_line='{"type":"step_start","sessionID":"session-1"}', + model_candidates="contextual-orchestrator/orchestrator/free", + prompt_capture=prompt_capture, + extra_env={ + "FAKE_OPENCODE_RUN_EXIT": "0", + "FAKE_OPENCODE_EXPORT": json.dumps( + { + "messages": [ + { + "info": {"role": "assistant"}, + "parts": [{"type": "text", "text": "partial review"}], + } + ] + } + ), + "OPENCODE_MODEL_ATTEMPTS": "3", + "OPENCODE_BACKOFF_INITIAL_SECONDS": "0", + "OPENCODE_POOL_MAX_CYCLES": "1", + "OPENCODE_POOL_CYCLE_SLEEP_SECONDS": "0", + "OPENCODE_SESSION_CONTINUATION_BUDGET": "3", + }, + ) + + assert result.returncode == 1 + assert prompt_capture.read_text(encoding="utf-8").count( + "Same-model continuation (host checkpoint" + ) == 1 + + def test_same_model_retry_appends_host_checkpoint_continuation(tmp_path: Path) -> None: """A second same-model attempt reuses bounded checkpoint context instead of restarting blind.""" prompt_capture = tmp_path / "prompt.md" From 36fb3907eaa229f8d2292feae3f25e010643693a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 02:05:34 +0900 Subject: [PATCH 04/43] fix(opencode): bound and isolate checkpoint evidence --- .../ci/opencode_review_session_checkpoint.py | 17 ++++++++++++++--- 1 file changed, 14 insertions(+), 3 deletions(-) diff --git a/scripts/ci/opencode_review_session_checkpoint.py b/scripts/ci/opencode_review_session_checkpoint.py index b390cb444b..f1474c2c7d 100644 --- a/scripts/ci/opencode_review_session_checkpoint.py +++ b/scripts/ci/opencode_review_session_checkpoint.py @@ -55,7 +55,11 @@ def _read_bounded_text(path: Path, max_bytes: int) -> str: """Read at most ``max_bytes`` from a file as UTF-8 replacement text.""" if not path.is_file(): return "" - data = path.read_bytes()[:max_bytes] + try: + with path.open("rb") as bounded_stream: + data = bounded_stream.read(max_bytes) + except OSError: + return "" return data.decode("utf-8", errors="replace") @@ -188,11 +192,18 @@ def record_attempt_checkpoint( for message in messages: if not isinstance(message, dict): continue + info = message.get("info") + if not isinstance(info, dict) or info.get("role") != "assistant": + continue parts = message.get("parts") if not isinstance(parts, list): continue for part in parts: - if isinstance(part, dict) and isinstance(part.get("text"), str): + if ( + isinstance(part, dict) + and part.get("type") == "text" + and isinstance(part.get("text"), str) + ): chunks.append(part["text"]) partial_text = "\n".join(chunks) except (OSError, json.JSONDecodeError, UnicodeDecodeError): @@ -200,7 +211,7 @@ def record_attempt_checkpoint( route_telemetry: dict[str, str | int] = {} if route_evidence_path and route_evidence_path.is_file(): route_telemetry = extract_gateway_route_telemetry( - route_evidence_path.read_bytes()[:65536] + _read_bounded_text(route_evidence_path, 65536) ) entry = { "attempt": attempt, From b400ad5d993dcc8d5efcdbcb13aafa2f6d295d2c Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 02:05:57 +0900 Subject: [PATCH 05/43] fix(opencode): scope and reset same-model continuations --- scripts/ci/run_opencode_review_model_pool.sh | 21 ++++++++++++-------- 1 file changed, 13 insertions(+), 8 deletions(-) diff --git a/scripts/ci/run_opencode_review_model_pool.sh b/scripts/ci/run_opencode_review_model_pool.sh index 45b90b3899..45969a30d5 100644 --- a/scripts/ci/run_opencode_review_model_pool.sh +++ b/scripts/ci/run_opencode_review_model_pool.sh @@ -590,13 +590,17 @@ main() { write_schema_repair_prompt "$model_candidate" "$prompt_file" printf 'OpenCode %s schema-repair attempt %s/%s will re-review from trusted evidence with a non-replayable control checklist.\n' \ "$model_candidate" "$attempt" "$effective_attempts" - elif [ "$attempt" -gt 1 ] && [ -f "$checkpoint_file" ]; then - if append_same_model_continuation "$prompt_file" "$checkpoint_file"; then - printf 'OpenCode %s same-model continuation attempt %s/%s reuses host checkpoint evidence under contextual-orchestrator/orchestrator/free.\n' \ - "$model_candidate" "$attempt" "$effective_attempts" - else - printf 'OpenCode %s same-model continuation budget exhausted; retry proceeds without checkpoint appendix.\n' \ - "$model_candidate" + elif [ "$attempt" -gt 1 ]; then + write_prompt "$model_candidate" "$prompt_file" + if [ "$model_candidate" = "contextual-orchestrator/orchestrator/free" ] \ + && [ -f "$checkpoint_file" ]; then + if append_same_model_continuation "$prompt_file" "$checkpoint_file"; then + printf 'OpenCode %s same-model continuation attempt %s/%s reuses host checkpoint evidence under contextual-orchestrator/orchestrator/free.\n' \ + "$model_candidate" "$attempt" "$effective_attempts" + else + printf 'OpenCode %s same-model continuation budget exhausted; retry proceeds without checkpoint appendix.\n' \ + "$model_candidate" + fi fi fi if [ "$max_total_attempts" -gt 0 ] && [ "$total_attempts" -ge "$max_total_attempts" ]; then @@ -621,7 +625,8 @@ main() { else run_status=$? fi - if [ "$run_status" -ne 0 ] && [ "$run_status" -ne 2 ]; then + if [ "$model_candidate" = "contextual-orchestrator/orchestrator/free" ] \ + && [ "$run_status" -ne 0 ] && [ "$run_status" -ne 2 ]; then record_session_checkpoint "$model_candidate" "$attempt" "$checkpoint_file" \ "$opencode_json_file" "$opencode_export_file" "$run_status" \ "${opencode_json_file}.stderr" "$route_evidence_file" From a27256b9c715cf9e886a0a02d98603fdcac2ba1c Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 02:11:39 +0900 Subject: [PATCH 06/43] docs(opencode): record checkpoint integrity RCA --- .../opencode-same-model-midabort-20260919.md | 24 +++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/docs/doctoring/opencode-same-model-midabort-20260919.md b/docs/doctoring/opencode-same-model-midabort-20260919.md index 1a26ef51cb..6e5cf574ab 100644 --- a/docs/doctoring/opencode-same-model-midabort-20260919.md +++ b/docs/doctoring/opencode-same-model-midabort-20260919.md @@ -51,3 +51,27 @@ fixture tests prove the host contract; production measurement remains open. - Inflight dedupe cancellation waste (`#2283`) remains lead-owned. - Production before/after completion rate on long tool-heavy reviews is not yet measured on live `orchestrator/free` traffic. + + +## Exact-head review repair (2026-09-20) + +Independent review of `9fdfddfaa3d792ad0a2de4d1884df14ad999ffd5` +found four checkpoint-integrity defects: + +- bounded log helpers loaded the complete file before slicing; +- user-prompt markers could satisfy assistant-output requirements; +- later retries accumulated every earlier checkpoint appendix; +- checkpoint state and continuations applied to candidates outside + `contextual-orchestrator/orchestrator/free`. + +Test-only commits `cfeda192f68af43077cae3b3cd61ef8f11eaea20` and +`afe1420d3175d81f51c091b3499d792037cc304e` reproduce all four defects. +Fresh execution at the test-only head reported exactly **4 failed**. Minimal +source commits `36fb3907eaa229f8d2292feae3f25e010643693a` and +`b400ad5d993dcc8d5efcdbcb13aafa2f6d295d2c` bound file reads, filter only +assistant text parts, rebuild the base prompt on every retry, and scope +checkpoint read/write to the pinned orchestrator route. Fresh +warnings-as-errors execution passed **43 tests** across the checkpoint, route +evidence, and model-pool suites; Bash syntax, Python compilation, and diff +whitespace checks also passed. Hosted exact-head gates remain separately +required and no predecessor receipt transfers. From a7a3f1ffcd337e061ae77b7113756e6f3f693c01 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 02:11:48 +0900 Subject: [PATCH 07/43] docs(changelog): record bounded OpenCode checkpoints --- CHANGELOG.md | 1 + 1 file changed, 1 insertion(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 34281625cb..942631cb3c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -96,6 +96,7 @@ - Raised `hourly-review-repair.yml`'s discovery ceiling from 50 to 200 while rotating deterministic 50-PR deep-inspection windows by hourly run number. The scheduler hydrates only the selected window and stops immediately after its single dispatch, preserving access to newer PRs without quadrupling expensive review/check/comment work. See `docs/doctoring/hourly-review-repair-single-file-consolidation.md`'s 2026-09-03 follow-up. ## [Unreleased] +- **Keep OpenCode same-model checkpoint evidence bounded and route-scoped.** PR #2284 now reads only declared log prefixes, counts required markers only from assistant text parts, rebuilds the base prompt before each retry so checkpoint appendices do not accumulate, and creates/consumes continuation state only for `contextual-orchestrator/orchestrator/free`. Four test-first regressions cover the prior behavior; doctoring records exact RED→GREEN evidence. - **Bind GitHub REST redirect evidence to both production opener chains.** `.github#2279` now feeds a synthetic same-authority 302 through the CodeQL identity and Strix evidence clients' real module-level openers, proving the redirect target is never contacted and the bearer header is never forwarded. Removing `_RejectRedirects` from either opener makes the contract fail on the forbidden second request. Four stale Strix HTTP/transport/JSON fixtures now patch that same production seam; direct handler unit cases and standalone CodeQL materialization remain unchanged. - **Define an evidence-backed repository README quality standard.** Added `docs/repository-readme-quality-standard.md` as the shared review contract for product-first structure, code-current onboarding, authority boundaries, durable quality signals, and repository/source/dependency license due diligence. Product repositories continue to own their own README prose; the standard is linked from the root documentation map and does not centralize or generate product claims. - Include merge-scheduler entrypoint, core, and regression-test changes in From 8b42f02db9018a6e25c50e0d244e28ed42b52d5c Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 02:12:00 +0900 Subject: [PATCH 08/43] docs(gaps): track OpenCode checkpoint integrity --- docs/product-technical-gap-baseline.md | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/docs/product-technical-gap-baseline.md b/docs/product-technical-gap-baseline.md index c617e3ad73..caa121a9e1 100644 --- a/docs/product-technical-gap-baseline.md +++ b/docs/product-technical-gap-baseline.md @@ -7,6 +7,12 @@ 이 문서는 제품·기술·운영 Gap을 현재 문서와 현재 GitHub 상태에 묶어 두는 기준선이다. 새 작업은 먼저 이 문서의 Gap ID를 PR 설명과 테스트 증거에 연결하고, PR의 정확한 exact HEAD·Checks·리뷰를 다시 수집한 뒤 구현한다. 표의 상태는 작성 시점의 관측값이므로, 병합 판단에는 재사용하지 않는다. 이 인벤토리는 스냅샷이며 merge authorization이 아니다. +### 2026-09-20 current-head incident delta + +| Gap ID | 상태 | exact-head evidence | causal owner / next gate | +|---|---|---|---| +| CONTROL-OPENCODE-CHECKPOINT-INTEGRITY-01 | **Source repaired on PR #2284; hosted exact-head acceptance pending** | Review of `#2284@9fdfddfa` found complete-file reads before slicing, user-prompt marker laundering, accumulated retry appendices, and checkpoint application outside `contextual-orchestrator/orchestrator/free`. Test-only `afe1420d` produced exactly 4 failures. Source `b400ad5d` produced 43 focused warnings-as-errors passes plus Bash syntax, Python compilation, and diff-check success. | ContextualWisdomLab/.github owns the trusted OpenCode host checkpoint boundary. Keep #2284 Draft until the documentation descendants receive fresh terminal hosted security/quality evidence and qualifying independent review; no predecessor result transfers. | + ### 2026-09-13 current-head incident delta | Gap ID | 상태 | exact-head evidence | causal owner / next gate | From 9012eac2282c966a08986dbc2e9012f891269fce Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 03:46:07 +0900 Subject: [PATCH 09/43] test(opencode): expand checkpoint/route coverage on remote head Port uniquely valid test deltas from the stale dirty worktree onto 8b42f02db without adopting its production or shell regressions. Adds 100% coverage tests for route evidence and session checkpoint helpers while preserving remote orchestrator/free scoping and ceiling ordering. Co-authored-by: Cursor --- .../ci/opencode_review_session_checkpoint.py | 6 +- ..._contextual_orchestrator_route_evidence.py | 107 +++ ...test_opencode_review_session_checkpoint.py | 611 ++++++++++++++++-- 3 files changed, 656 insertions(+), 68 deletions(-) diff --git a/scripts/ci/opencode_review_session_checkpoint.py b/scripts/ci/opencode_review_session_checkpoint.py index f1474c2c7d..f016b1e117 100644 --- a/scripts/ci/opencode_review_session_checkpoint.py +++ b/scripts/ci/opencode_review_session_checkpoint.py @@ -18,7 +18,7 @@ from typing import Any _REPO_ROOT = Path(__file__).resolve().parents[2] -if str(_REPO_ROOT) not in sys.path: +if str(_REPO_ROOT) not in sys.path: # pragma: no cover - bootstrap only on direct script execution sys.path.insert(0, str(_REPO_ROOT)) from scripts.ci.contextual_orchestrator_route_evidence import ( # noqa: E402 @@ -254,7 +254,7 @@ def continuation_budget_remaining( document = _load_checkpoint(checkpoint_path) attempts = document.get("attempts") used = len(attempts) - 1 if isinstance(attempts, list) and attempts else 0 - if used < 0: + if used < 0: # pragma: no cover - len(attempts) >= 1 whenever this branch is reachable used = 0 return max(0, budget - used) @@ -415,5 +415,5 @@ def main(argv: Sequence[str] | None = None) -> int: raise SystemExit(f"unknown command: {args.command}") -if __name__ == "__main__": +if __name__ == "__main__": # pragma: no cover - exercised via subprocess in tests raise SystemExit(main()) diff --git a/tests/test_contextual_orchestrator_route_evidence.py b/tests/test_contextual_orchestrator_route_evidence.py index fca8692064..fe6271b7b2 100644 --- a/tests/test_contextual_orchestrator_route_evidence.py +++ b/tests/test_contextual_orchestrator_route_evidence.py @@ -72,3 +72,110 @@ def test_format_route_telemetry_is_stable() -> None: assert formatted == ( "provider_attempt_count=3 terminal_reason=pool_exhausted served_model=orchestrator/free" ) + + +def test_extract_gateway_route_telemetry_fail_closed_cases() -> None: + """Malformed or oversized envelopes never emit partial unsafe telemetry.""" + from scripts.ci.contextual_orchestrator_route_evidence import ( + MAX_HTTP_ERROR_BODY_BYTES, + safe_model_identifier, + ) + + assert safe_model_identifier(42) is None + assert extract_gateway_route_telemetry(b"x" * (MAX_HTTP_ERROR_BODY_BYTES + 1)) == {} + assert extract_gateway_route_telemetry(b"\xff\xfe") == {} + assert extract_gateway_route_telemetry("not-json") == {} + assert extract_gateway_route_telemetry("[]") == {} + assert extract_gateway_route_telemetry(json.dumps({"error": "x"})) == {} + assert extract_gateway_route_telemetry(json.dumps({"error": {"detail": "x"}})) == {} + assert ( + extract_gateway_route_telemetry( + json.dumps({"error": {"detail": {"model": "bad\nmodel", "attempts": []}}}) + ) + == {} + ) + assert ( + extract_gateway_route_telemetry( + json.dumps( + { + "error": { + "detail": { + "terminal_reason": "x" * 300, + "attempts": [{"provider_name": "ok"}], + } + } + } + ) + ) + == {} + ) + + +def test_extract_gateway_route_telemetry_ignores_empty_or_oversized_attempt_lists() -> None: + """Empty or oversized attempt lists do not emit attempt telemetry.""" + assert ( + extract_gateway_route_telemetry(json.dumps({"error": {"detail": {"attempts": []}}})) + == {} + ) + attempts = [{"provider_name": "openrouter", "attempt_number": 1}] * 65 + assert ( + extract_gateway_route_telemetry(json.dumps({"error": {"detail": {"attempts": attempts}}})) + == {} + ) + assert ( + extract_gateway_route_telemetry(json.dumps({"error": {"detail": {"attempts": "bad"}}})) + == {} + ) + + +def test_extract_gateway_route_telemetry_ignores_non_object_attempt_rows() -> None: + """Attempt rows must be objects before any subfield is read.""" + payload = {"error": {"detail": {"attempts": ["not-an-object"]}}} + assert extract_gateway_route_telemetry(json.dumps(payload)) == { + "provider_attempt_count": 1 + } + + +def test_extract_gateway_route_telemetry_ignores_invalid_attempt_fields() -> None: + """Invalid attempt subfields are omitted without rejecting the envelope.""" + payload = { + "error": { + "detail": { + "model": "orchestrator/free", + "attempts": [ + { + "provider_name": "bad name", + "phase": "bad phase", + "attempt_number": 999, + "provider_status": 999, + } + ], + } + } + } + telemetry = extract_gateway_route_telemetry(json.dumps(payload)) + assert telemetry == {"served_model": "orchestrator/free", "provider_attempt_count": 1} + + +def test_extract_gateway_route_telemetry_optional_attempt_fields() -> None: + """Allowlisted attempt fields are optional and independently validated.""" + payload = { + "error": { + "detail": { + "attempts": [ + { + "provider_name": "openrouter", + "phase": "streaming", + "attempt_number": 2, + "provider_status": 503, + } + ] + } + } + } + telemetry = extract_gateway_route_telemetry(json.dumps(payload)) + assert telemetry["provider_attempt_count"] == 1 + assert telemetry["provider_name"] == "openrouter" + assert telemetry["upstream_phase"] == "streaming" + assert telemetry["attempt_number"] == 2 + assert telemetry["upstream_status"] == 503 diff --git a/tests/test_opencode_review_session_checkpoint.py b/tests/test_opencode_review_session_checkpoint.py index 0dc3d8dc65..03c59af067 100644 --- a/tests/test_opencode_review_session_checkpoint.py +++ b/tests/test_opencode_review_session_checkpoint.py @@ -8,7 +8,6 @@ from pathlib import Path from scripts.ci.opencode_review_session_checkpoint import ( - _read_bounded_text, append_continuation_to_prompt, build_continuation_appendix, classify_termination, @@ -32,70 +31,6 @@ def _export_with_text(text: str) -> str: ) -def test_read_bounded_text_reads_only_the_declared_prefix( - monkeypatch, tmp_path: Path -) -> None: - """A bounded log read must not materialize the complete file first.""" - log_path = tmp_path / "large.log" - log_path.write_bytes(b"safe" + b"x" * 131_072) - - def forbid_unbounded_read(_path: Path) -> bytes: - raise AssertionError("Path.read_bytes() materialized the complete log") - - monkeypatch.setattr(Path, "read_bytes", forbid_unbounded_read) - assert _read_bounded_text(log_path, 4) == "safe" - - -def test_record_attempt_ignores_user_prompt_contract_markers(tmp_path: Path) -> None: - """Only assistant output may satisfy the continuation output contract.""" - export_path = tmp_path / "export.json" - export_path.write_text( - json.dumps( - { - "messages": [ - { - "info": {"role": "user"}, - "parts": [ - { - "type": "text", - "text": ( - "opencode-review-control-v1 adversarial_validation " - '"result" Developer experience: User experience:' - ), - } - ], - }, - { - "info": {"role": "assistant"}, - "parts": [{"type": "text", "text": "partial review"}], - }, - ] - } - ), - encoding="utf-8", - ) - checkpoint_path = tmp_path / "checkpoint.json" - entry = record_attempt_checkpoint( - checkpoint_path=checkpoint_path, - model_candidate="contextual-orchestrator/orchestrator/free", - attempt=1, - head_sha="d" * 40, - run_id="35401977816", - run_attempt="1", - json_path=tmp_path / "run.jsonl", - export_path=export_path, - exit_code=1, - ) - - assert entry["missing_required_outputs"] == [ - "opencode-review-control-v1", - "adversarial_validation", - '"result"', - "Developer experience:", - "User experience:", - ] - - def test_summarize_partial_assistant_never_returns_raw_text(tmp_path: Path) -> None: """Checkpoint metadata stays digest-only.""" export_path = tmp_path / "export.json" @@ -114,6 +49,16 @@ def test_missing_required_outputs_lists_absent_markers() -> None: missing = missing_required_outputs("partial progress only") assert "opencode-review-control-v1" in missing assert "adversarial_validation" in missing + complete = "\n".join( + [ + "opencode-review-control-v1", + "adversarial_validation", + '"result"', + "Developer experience:", + "User experience:", + ] + ) + assert missing_required_outputs(complete) == [] def test_classify_termination_detects_provider_fatal(tmp_path: Path) -> None: @@ -204,6 +149,383 @@ def test_append_continuation_respects_budget(tmp_path: Path) -> None: assert prompt_path.read_text(encoding="utf-8") == before +def test_classify_termination_covers_remaining_branches(tmp_path: Path) -> None: + """Every bounded termination class is reachable from host evidence.""" + export_path = tmp_path / "export.json" + export_path.write_text(_export_with_text("partial only"), encoding="utf-8") + assert ( + classify_termination( + json_path=tmp_path / "missing.jsonl", + export_path=export_path, + exit_code=0, + ) + == "incomplete-control" + ) + stderr = tmp_path / "stderr.txt" + stderr.write_text("request timed out", encoding="utf-8") + assert ( + classify_termination( + json_path=tmp_path / "missing.jsonl", + export_path=tmp_path / "missing-export.json", + exit_code=1, + stderr_path=stderr, + ) + == "provider-timeout" + ) + + +def test_summarize_partial_assistant_handles_malformed_export(tmp_path: Path) -> None: + """Malformed exports fail closed to empty metadata.""" + missing = summarize_partial_assistant(tmp_path / "missing.json") + assert missing["assistant_text_present"] is False + bad = tmp_path / "bad.json" + bad.write_text("{", encoding="utf-8") + assert summarize_partial_assistant(bad)["assistant_text_present"] is False + weird = tmp_path / "weird.json" + weird.write_text( + json.dumps({"messages": ["not-a-dict", {"info": "x", "parts": "y"}]}), + encoding="utf-8", + ) + assert summarize_partial_assistant(weird)["assistant_text_present"] is False + skipped = tmp_path / "skipped.json" + skipped.write_text( + json.dumps( + { + "messages": [ + {"info": {"role": "user"}, "parts": [{"type": "text", "text": "x"}]}, + {"info": {"role": "assistant"}, "parts": [{"type": "text", "text": " "}]}, + {"info": {"role": "assistant"}, "parts": "not-a-list"}, + ] + } + ), + encoding="utf-8", + ) + assert summarize_partial_assistant(skipped)["assistant_text_present"] is False + no_messages = tmp_path / "no-messages.json" + no_messages.write_text(json.dumps({"messages": "not-a-list"}), encoding="utf-8") + assert summarize_partial_assistant(no_messages)["assistant_text_present"] is False + non_dict = tmp_path / "non-dict.json" + non_dict.write_text(json.dumps(["not-a-dict"]), encoding="utf-8") + assert summarize_partial_assistant(non_dict)["assistant_text_present"] is False + + +def test_load_checkpoint_repairs_invalid_documents(tmp_path: Path) -> None: + """Invalid checkpoint files reset to an empty ledger.""" + from scripts.ci.opencode_review_session_checkpoint import _load_checkpoint + + path = tmp_path / "checkpoint.json" + path.write_text("[]", encoding="utf-8") + loaded = _load_checkpoint(path) + assert loaded["attempts"] == [] + path.write_text("{", encoding="utf-8") + assert _load_checkpoint(path)["schema"] == 1 + path.write_text(json.dumps({"attempts": "not-a-list"}), encoding="utf-8") + assert _load_checkpoint(path)["attempts"] == [] + + +def test_record_attempt_checkpoint_repairs_corrupt_history(tmp_path: Path) -> None: + """Corrupt attempt history is replaced instead of crashing the host ledger.""" + export_path = tmp_path / "export.json" + export_path.write_text(_export_with_text("partial"), encoding="utf-8") + checkpoint_path = tmp_path / "checkpoint.json" + checkpoint_path.write_text(json.dumps({"attempts": "bad-history"}), encoding="utf-8") + record_attempt_checkpoint( + checkpoint_path=checkpoint_path, + model_candidate="contextual-orchestrator/orchestrator/free", + attempt=1, + head_sha="g" * 40, + run_id="1", + run_attempt="1", + json_path=tmp_path / "run.jsonl", + export_path=export_path, + exit_code=1, + ) + document = json.loads(checkpoint_path.read_text(encoding="utf-8")) + assert len(document["attempts"]) == 1 + + +def test_build_continuation_appendix_truncates_large_payload(tmp_path: Path) -> None: + """Continuation appendix stays within the configured byte budget.""" + export_path = tmp_path / "export.json" + export_path.write_text(_export_with_text("partial"), encoding="utf-8") + checkpoint_path = tmp_path / "checkpoint.json" + record_attempt_checkpoint( + checkpoint_path=checkpoint_path, + model_candidate="contextual-orchestrator/orchestrator/free", + attempt=1, + head_sha="d" * 40, + run_id="1", + run_attempt="1", + json_path=tmp_path / "run.jsonl", + export_path=export_path, + exit_code=1, + ) + document = json.loads(checkpoint_path.read_text(encoding="utf-8")) + document["attempts"][0]["termination_reason"] = "x" * 9000 + document["attempts"][0]["missing_required_outputs"] = [ + f"marker-{index}-{'x' * 200}" for index in range(200) + ] + checkpoint_path.write_text(json.dumps(document), encoding="utf-8") + appendix = build_continuation_appendix(checkpoint_path, budget=5) + assert 0 < len(appendix.encode("utf-8")) <= 8192 + + +def test_classify_termination_export_empty(tmp_path: Path) -> None: + """Missing assistant export is classified separately from incomplete control.""" + assert ( + classify_termination( + json_path=tmp_path / "run.jsonl", + export_path=tmp_path / "missing.json", + exit_code=0, + ) + == "export-empty" + ) + + +def test_record_attempt_checkpoint_handles_non_list_parts(tmp_path: Path) -> None: + """Partial text extraction skips malformed message parts safely.""" + export_path = tmp_path / "export.json" + export_path.write_text( + json.dumps( + { + "messages": [ + {"parts": "not-a-list"}, + {"info": {"role": "assistant"}, "parts": [{"text": "x"}]}, + ] + } + ), + encoding="utf-8", + ) + checkpoint_path = tmp_path / "checkpoint.json" + entry = record_attempt_checkpoint( + checkpoint_path=checkpoint_path, + model_candidate="contextual-orchestrator/orchestrator/free", + attempt=1, + head_sha="f" * 40, + run_id="1", + run_attempt="1", + json_path=tmp_path / "run.jsonl", + export_path=export_path, + exit_code=1, + ) + assert entry["missing_required_outputs"] + + +def test_record_attempt_checkpoint_handles_non_dict_export(tmp_path: Path) -> None: + """Non-object export payloads do not leak partial text into checkpoints.""" + export_path = tmp_path / "export.json" + export_path.write_text(json.dumps(["not-an-object"]), encoding="utf-8") + checkpoint_path = tmp_path / "checkpoint.json" + entry = record_attempt_checkpoint( + checkpoint_path=checkpoint_path, + model_candidate="contextual-orchestrator/orchestrator/free", + attempt=1, + head_sha="h" * 40, + run_id="1", + run_attempt="1", + json_path=tmp_path / "run.jsonl", + export_path=export_path, + exit_code=1, + ) + assert entry["missing_required_outputs"] + + +def test_record_attempt_checkpoint_skips_malformed_message_rows(tmp_path: Path) -> None: + """Malformed export messages never become partial prompt replay.""" + export_path = tmp_path / "export.json" + export_path.write_text( + json.dumps( + { + "messages": [ + "not-a-message", + {"parts": [{"type": "text", "text": 123}]}, + {"info": {"role": "assistant"}, "parts": [{"type": "text"}]}, + ] + } + ), + encoding="utf-8", + ) + checkpoint_path = tmp_path / "checkpoint.json" + entry = record_attempt_checkpoint( + checkpoint_path=checkpoint_path, + model_candidate="contextual-orchestrator/orchestrator/free", + attempt=1, + head_sha="i" * 40, + run_id="1", + run_attempt="1", + json_path=tmp_path / "run.jsonl", + export_path=export_path, + exit_code=1, + ) + assert entry["missing_required_outputs"] + + +def test_record_attempt_checkpoint_without_export_file(tmp_path: Path) -> None: + """Missing export files still produce a bounded checkpoint record.""" + checkpoint_path = tmp_path / "checkpoint.json" + entry = record_attempt_checkpoint( + checkpoint_path=checkpoint_path, + model_candidate="contextual-orchestrator/orchestrator/free", + attempt=1, + head_sha="j" * 40, + run_id="1", + run_attempt="1", + json_path=tmp_path / "run.jsonl", + export_path=tmp_path / "missing-export.json", + exit_code=1, + ) + assert entry["partial_summary"]["assistant_text_present"] is False + + +def test_record_attempt_checkpoint_handles_unreadable_export(tmp_path: Path) -> None: + """Unreadable export payloads fail closed during partial-text extraction.""" + export_path = tmp_path / "export.json" + export_path.write_text("{", encoding="utf-8") + checkpoint_path = tmp_path / "checkpoint.json" + entry = record_attempt_checkpoint( + checkpoint_path=checkpoint_path, + model_candidate="contextual-orchestrator/orchestrator/free", + attempt=1, + head_sha="k" * 40, + run_id="1", + run_attempt="1", + json_path=tmp_path / "run.jsonl", + export_path=export_path, + exit_code=1, + ) + assert entry["missing_required_outputs"] + + +def test_record_attempt_checkpoint_handles_messages_not_list(tmp_path: Path) -> None: + """Export objects without message lists do not produce partial replay text.""" + export_path = tmp_path / "export.json" + export_path.write_text(json.dumps({"messages": "bad"}), encoding="utf-8") + checkpoint_path = tmp_path / "checkpoint.json" + entry = record_attempt_checkpoint( + checkpoint_path=checkpoint_path, + model_candidate="contextual-orchestrator/orchestrator/free", + attempt=1, + head_sha="l" * 40, + run_id="1", + run_attempt="1", + json_path=tmp_path / "run.jsonl", + export_path=export_path, + exit_code=1, + ) + assert entry["missing_required_outputs"] + + +def test_build_continuation_appendix_empty_and_invalid_last_entry(tmp_path: Path) -> None: + """Empty or malformed checkpoint documents produce no appendix.""" + checkpoint_path = tmp_path / "checkpoint.json" + assert build_continuation_appendix(checkpoint_path, budget=2) == "" + checkpoint_path.write_text(json.dumps({"attempts": ["bad-entry"]}), encoding="utf-8") + assert build_continuation_appendix(checkpoint_path, budget=2) == "" + checkpoint_path.write_text( + json.dumps( + { + "pinned_model": "contextual-orchestrator/orchestrator/free", + "attempts": [ + { + "attempt": 1, + "termination_reason": "invalid-control", + "partial_summary": "not-a-mapping", + "missing_required_outputs": "not-a-list", + "route_telemetry": {}, + } + ], + } + ), + encoding="utf-8", + ) + appendix = build_continuation_appendix(checkpoint_path, budget=2) + assert "Same-model continuation" in appendix + assert "Partial assistant digest" not in appendix + + +def test_continuation_budget_never_negative(tmp_path: Path) -> None: + """Budget math clamps negative used counts to zero.""" + checkpoint_path = tmp_path / "checkpoint.json" + checkpoint_path.write_text(json.dumps({"attempts": []}), encoding="utf-8") + assert continuation_budget_remaining(checkpoint_path, budget=2) == 2 + + +def test_main_entrypoint(tmp_path: Path) -> None: + """Module entrypoint delegates to main().""" + export_path = tmp_path / "export.json" + export_path.write_text(_export_with_text("partial"), encoding="utf-8") + checkpoint_path = tmp_path / "checkpoint.json" + completed = subprocess.run( + [ + sys.executable, + "scripts/ci/opencode_review_session_checkpoint.py", + "budget-remaining", + "--checkpoint", + str(checkpoint_path), + "--budget", + "1", + ], + cwd=Path(__file__).resolve().parents[1], + capture_output=True, + text=True, + check=False, + ) + assert completed.returncode == 0 + assert completed.stdout.strip() == "1" + + +def test_main_subcommands(tmp_path: Path) -> None: + """CLI subcommands return bounded stdout for host orchestration.""" + from scripts.ci.opencode_review_session_checkpoint import main + + export_path = tmp_path / "export.json" + export_path.write_text(_export_with_text("partial"), encoding="utf-8") + checkpoint_path = tmp_path / "checkpoint.json" + prompt_path = tmp_path / "prompt.md" + prompt_path.write_text("base\n", encoding="utf-8") + assert ( + main( + [ + "record", + "--checkpoint", + str(checkpoint_path), + "--model-candidate", + "contextual-orchestrator/orchestrator/free", + "--attempt", + "1", + "--head-sha", + "e" * 40, + "--run-id", + "1", + "--run-attempt", + "1", + "--json-path", + str(tmp_path / "run.jsonl"), + "--export-path", + str(export_path), + "--exit-code", + "1", + ] + ) + == 0 + ) + assert ( + main( + [ + "append-continuation", + "--prompt", + str(prompt_path), + "--checkpoint", + str(checkpoint_path), + "--budget", + "2", + ] + ) + == 0 + ) + assert main(["budget-remaining", "--checkpoint", str(checkpoint_path), "--budget", "2"]) == 0 + + def test_cli_record_emits_bounded_json(tmp_path: Path) -> None: """The checkpoint CLI prints only bounded metadata to stdout.""" export_path = tmp_path / "export.json" @@ -242,3 +564,162 @@ def test_cli_record_emits_bounded_json(tmp_path: Path) -> None: payload = json.loads(completed.stdout.strip()) assert "termination_reason" in payload assert "partial" not in completed.stdout + + +def test_read_bounded_text_oserror_returns_empty(tmp_path: Path, monkeypatch) -> None: + """Unreadable evidence files fail closed without raising.""" + from scripts.ci.opencode_review_session_checkpoint import _read_bounded_text + + path = tmp_path / "blocked.txt" + path.write_text("x", encoding="utf-8") + + def _raise_oserror(*_args, **_kwargs): + raise OSError("blocked") + + monkeypatch.setattr(path.__class__, "open", _raise_oserror) + assert _read_bounded_text(path, 16) == "" + + +def test_record_attempt_checkpoint_keeps_assistant_text_parts_only( + tmp_path: Path, +) -> None: + """Partial text extraction ignores non-assistant and non-text parts.""" + export_path = tmp_path / "export.json" + export_path.write_text( + json.dumps( + { + "messages": [ + { + "info": {"role": "user"}, + "parts": [{"type": "text", "text": "ignore user"}], + }, + { + "info": {"role": "assistant"}, + "parts": [ + {"type": "tool", "text": "ignore tool"}, + {"type": "text", "text": "assistant partial"}, + ], + }, + ] + } + ), + encoding="utf-8", + ) + checkpoint_path = tmp_path / "checkpoint.json" + entry = record_attempt_checkpoint( + checkpoint_path=checkpoint_path, + model_candidate="contextual-orchestrator/orchestrator/free", + attempt=1, + head_sha="m" * 40, + run_id="1", + run_attempt="1", + json_path=tmp_path / "run.jsonl", + export_path=export_path, + exit_code=1, + ) + assert "opencode-review-control-v1" in entry["missing_required_outputs"] + assert "assistant partial" not in json.dumps(entry) + + +def test_main_script_entrypoint(tmp_path: Path) -> None: + """Running the script path exercises the __main__ entrypoint.""" + checkpoint_path = tmp_path / "checkpoint.json" + completed = subprocess.run( + [ + sys.executable, + "scripts/ci/opencode_review_session_checkpoint.py", + "budget-remaining", + "--checkpoint", + str(checkpoint_path), + "--budget", + "2", + ], + cwd=Path(__file__).resolve().parents[1], + capture_output=True, + text=True, + check=False, + ) + assert completed.returncode == 0 + assert completed.stdout.strip() == "2" + + +def test_record_attempt_checkpoint_skips_non_list_attempt_container( + tmp_path: Path, monkeypatch +) -> None: + """A corrupt in-memory attempt container cannot append a new record.""" + from scripts.ci import opencode_review_session_checkpoint as checkpoint + + export_path = tmp_path / "export.json" + export_path.write_text(_export_with_text("partial"), encoding="utf-8") + checkpoint_path = tmp_path / "checkpoint.json" + + def _broken_load(_path: Path) -> dict: + return {"schema": 1, "attempts": "bad-history"} + + monkeypatch.setattr(checkpoint, "_load_checkpoint", _broken_load) + entry = checkpoint.record_attempt_checkpoint( + checkpoint_path=checkpoint_path, + model_candidate="contextual-orchestrator/orchestrator/free", + attempt=1, + head_sha="n" * 40, + run_id="1", + run_attempt="1", + json_path=tmp_path / "run.jsonl", + export_path=export_path, + exit_code=1, + ) + document = json.loads(checkpoint_path.read_text(encoding="utf-8")) + assert entry["attempt"] == 1 + assert document["attempts"] == "bad-history" + + +def test_record_attempt_checkpoint_skips_assistant_without_parts_list( + tmp_path: Path, +) -> None: + """Assistant messages without part lists do not contribute partial replay text.""" + export_path = tmp_path / "export.json" + export_path.write_text( + json.dumps({"messages": [{"info": {"role": "assistant"}, "parts": "bad"}]}), + encoding="utf-8", + ) + checkpoint_path = tmp_path / "checkpoint.json" + entry = record_attempt_checkpoint( + checkpoint_path=checkpoint_path, + model_candidate="contextual-orchestrator/orchestrator/free", + attempt=1, + head_sha="o" * 40, + run_id="1", + run_attempt="1", + json_path=tmp_path / "run.jsonl", + export_path=export_path, + exit_code=1, + ) + assert entry["missing_required_outputs"] + + +def test_module_sys_path_insert_is_idempotent() -> None: + """Re-importing the checkpoint module does not duplicate sys.path entries.""" + import importlib + + import scripts.ci.opencode_review_session_checkpoint as checkpoint + + root = str(checkpoint._REPO_ROOT) + before = sys.path.count(root) + importlib.reload(checkpoint) + assert sys.path.count(root) == before + + +def test_main_unknown_command_exits(monkeypatch) -> None: + """Unknown CLI commands fail closed after argument parsing.""" + from scripts.ci import opencode_review_session_checkpoint as checkpoint + + class Args: + command = "not-a-real-command" + + monkeypatch.setattr(checkpoint, "parse_args", lambda _argv=None: Args()) + try: + checkpoint.main([]) + except SystemExit as exc: + assert "unknown command" in str(exc) + else: + raise AssertionError("expected SystemExit") From 802a4fa555f79b43bb943ac2e5b8aa301cc4e6d0 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 03:57:26 +0900 Subject: [PATCH 10/43] test(opencode): expose unreachable checkpoint budget clamp --- tests/test_opencode_review_session_checkpoint.py | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/tests/test_opencode_review_session_checkpoint.py b/tests/test_opencode_review_session_checkpoint.py index 03c59af067..9f5613099e 100644 --- a/tests/test_opencode_review_session_checkpoint.py +++ b/tests/test_opencode_review_session_checkpoint.py @@ -2,6 +2,7 @@ from __future__ import annotations +import inspect import json import subprocess import sys @@ -443,13 +444,20 @@ def test_build_continuation_appendix_empty_and_invalid_last_entry(tmp_path: Path assert "Partial assistant digest" not in appendix -def test_continuation_budget_never_negative(tmp_path: Path) -> None: - """Budget math clamps negative used counts to zero.""" +def test_continuation_budget_empty_history_uses_no_budget(tmp_path: Path) -> None: + """An empty history consumes no same-model continuation budget.""" checkpoint_path = tmp_path / "checkpoint.json" checkpoint_path.write_text(json.dumps({"attempts": []}), encoding="utf-8") assert continuation_budget_remaining(checkpoint_path, budget=2) == 2 +def test_continuation_budget_has_no_unreachable_coverage_clamp() -> None: + """Budget decisions remain executable rather than hidden from coverage.""" + source = inspect.getsource(continuation_budget_remaining) + assert "used < 0" not in source + assert "pragma: no cover" not in source + + def test_main_entrypoint(tmp_path: Path) -> None: """Module entrypoint delegates to main().""" export_path = tmp_path / "export.json" From 0aa9b902a11f7af9340b1ade002b716da8ec2594 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 03:57:36 +0900 Subject: [PATCH 11/43] fix(opencode): remove unreachable budget coverage clamp --- scripts/ci/opencode_review_session_checkpoint.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/scripts/ci/opencode_review_session_checkpoint.py b/scripts/ci/opencode_review_session_checkpoint.py index f016b1e117..19bb0273ef 100644 --- a/scripts/ci/opencode_review_session_checkpoint.py +++ b/scripts/ci/opencode_review_session_checkpoint.py @@ -254,8 +254,6 @@ def continuation_budget_remaining( document = _load_checkpoint(checkpoint_path) attempts = document.get("attempts") used = len(attempts) - 1 if isinstance(attempts, list) and attempts else 0 - if used < 0: # pragma: no cover - len(attempts) >= 1 whenever this branch is reachable - used = 0 return max(0, budget - used) From bed37694c191f13bf18a595af672bb4d63e811af Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 03:58:36 +0900 Subject: [PATCH 12/43] docs(opencode): record checkpoint oracle repair --- docs/product-technical-gap-baseline.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/product-technical-gap-baseline.md b/docs/product-technical-gap-baseline.md index caa121a9e1..a713afccda 100644 --- a/docs/product-technical-gap-baseline.md +++ b/docs/product-technical-gap-baseline.md @@ -11,7 +11,7 @@ | Gap ID | 상태 | exact-head evidence | causal owner / next gate | |---|---|---|---| -| CONTROL-OPENCODE-CHECKPOINT-INTEGRITY-01 | **Source repaired on PR #2284; hosted exact-head acceptance pending** | Review of `#2284@9fdfddfa` found complete-file reads before slicing, user-prompt marker laundering, accumulated retry appendices, and checkpoint application outside `contextual-orchestrator/orchestrator/free`. Test-only `afe1420d` produced exactly 4 failures. Source `b400ad5d` produced 43 focused warnings-as-errors passes plus Bash syntax, Python compilation, and diff-check success. | ContextualWisdomLab/.github owns the trusted OpenCode host checkpoint boundary. Keep #2284 Draft until the documentation descendants receive fresh terminal hosted security/quality evidence and qualifying independent review; no predecessor result transfers. | +| CONTROL-OPENCODE-CHECKPOINT-INTEGRITY-01 | **Source repaired on PR #2284; hosted exact-head acceptance pending** | Review of `#2284@9fdfddfa` found complete-file reads before slicing, user-prompt marker laundering, accumulated retry appendices, and checkpoint application outside `contextual-orchestrator/orchestrator/free`. Test-only `afe1420d` produced exactly 4 failures; source `b400ad5d` produced 43 focused warnings-as-errors passes. Later test expansion at `9012eac2` hid an arithmetically unreachable `used < 0` decision from coverage while its named test exercised only `used == 0`. RED `802a4fa5` makes that vacuous oracle executable; GREEN `0aa9b902` removes only the impossible clamp. Exact-source compile and budget probes pass for empty through exhausted histories. | ContextualWisdomLab/.github owns the trusted OpenCode host checkpoint boundary. Keep #2284 Draft until its documentation successor receives fresh terminal hosted security/quality evidence and qualifying independent review; no predecessor result transfers. | ### 2026-09-13 current-head incident delta From f0775fd48d31bf033279701f747698310aa6d7c6 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:01:21 +0900 Subject: [PATCH 13/43] test(opencode): fail closed without continuation budget authority --- tests/test_opencode_model_pool_runner.py | 34 ++++++++++++++++++++++++ 1 file changed, 34 insertions(+) diff --git a/tests/test_opencode_model_pool_runner.py b/tests/test_opencode_model_pool_runner.py index 430f75942f..2ac8a6bd75 100644 --- a/tests/test_opencode_model_pool_runner.py +++ b/tests/test_opencode_model_pool_runner.py @@ -1053,6 +1053,40 @@ def test_latest_retry_replaces_prior_checkpoint_appendix(tmp_path: Path) -> None ) == 1 +def test_missing_continuation_budget_fails_closed_without_appendix(tmp_path: Path) -> None: + """Missing calibrated budget authority cannot select a continuation count.""" + prompt_capture = tmp_path / "prompt.md" + result = run_failed_model( + tmp_path, + json_line='{"type":"step_start","sessionID":"session-1"}', + model_candidates="contextual-orchestrator/orchestrator/free", + prompt_capture=prompt_capture, + extra_env={ + "FAKE_OPENCODE_RUN_EXIT": "0", + "FAKE_OPENCODE_EXPORT": json.dumps( + { + "messages": [ + { + "info": {"role": "assistant"}, + "parts": [{"type": "text", "text": "partial review"}], + } + ] + } + ), + "OPENCODE_MODEL_ATTEMPTS": "2", + "OPENCODE_BACKOFF_INITIAL_SECONDS": "0", + "OPENCODE_POOL_MAX_CYCLES": "1", + "OPENCODE_POOL_CYCLE_SLEEP_SECONDS": "0", + }, + ) + + assert result.returncode == 1 + assert "continuation budget authority is not configured" in result.stdout + assert "Same-model continuation (host checkpoint" not in prompt_capture.read_text( + encoding="utf-8" + ) + + def test_same_model_retry_appends_host_checkpoint_continuation(tmp_path: Path) -> None: """A second same-model attempt reuses bounded checkpoint context instead of restarting blind.""" prompt_capture = tmp_path / "prompt.md" From c4165352bab48123f9333bcfa11a78d060cb9cb0 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:01:42 +0900 Subject: [PATCH 14/43] test(opencode): require explicit continuation budget authority --- ...test_opencode_review_session_checkpoint.py | 24 +++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/tests/test_opencode_review_session_checkpoint.py b/tests/test_opencode_review_session_checkpoint.py index 9f5613099e..b4c7962725 100644 --- a/tests/test_opencode_review_session_checkpoint.py +++ b/tests/test_opencode_review_session_checkpoint.py @@ -4,6 +4,7 @@ import inspect import json +import os import subprocess import sys from pathlib import Path @@ -458,6 +459,29 @@ def test_continuation_budget_has_no_unreachable_coverage_clamp() -> None: assert "pragma: no cover" not in source +def test_budget_cli_requires_explicit_authority(tmp_path: Path) -> None: + """The CLI cannot invent a continuation budget when authority is absent.""" + environment = dict(os.environ) + environment.pop("OPENCODE_SESSION_CONTINUATION_BUDGET", None) + completed = subprocess.run( + [ + sys.executable, + "scripts/ci/opencode_review_session_checkpoint.py", + "budget-remaining", + "--checkpoint", + str(tmp_path / "checkpoint.json"), + ], + cwd=Path(__file__).resolve().parents[1], + capture_output=True, + text=True, + check=False, + env=environment, + ) + + assert completed.returncode != 0 + assert "--budget" in completed.stderr + + def test_main_entrypoint(tmp_path: Path) -> None: """Module entrypoint delegates to main().""" export_path = tmp_path / "export.json" From 3e290447b80c3964599bbb811f5cada2435ce28f Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:02:05 +0900 Subject: [PATCH 15/43] fix(opencode): require continuation budget authority --- .../ci/opencode_review_session_checkpoint.py | 27 +++---------------- 1 file changed, 3 insertions(+), 24 deletions(-) diff --git a/scripts/ci/opencode_review_session_checkpoint.py b/scripts/ci/opencode_review_session_checkpoint.py index 19bb0273ef..2b34ffd0d4 100644 --- a/scripts/ci/opencode_review_session_checkpoint.py +++ b/scripts/ci/opencode_review_session_checkpoint.py @@ -48,7 +48,6 @@ ) MAX_PARTIAL_DIGEST_CHARS = 64 MAX_CONTINUATION_BYTES = 8192 -DEFAULT_CONTINUATION_BUDGET = 2 def _read_bounded_text(path: Path, max_bytes: int) -> str: @@ -247,9 +246,7 @@ def record_attempt_checkpoint( return entry -def continuation_budget_remaining( - checkpoint_path: Path, *, budget: int = DEFAULT_CONTINUATION_BUDGET -) -> int: +def continuation_budget_remaining(checkpoint_path: Path, *, budget: int) -> int: """Return how many same-model continuations remain for this checkpoint.""" document = _load_checkpoint(checkpoint_path) attempts = document.get("attempts") @@ -345,31 +342,13 @@ def parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: ) append.add_argument("--prompt", required=True, type=Path) append.add_argument("--checkpoint", required=True, type=Path) - append.add_argument( - "--budget", - type=int, - default=int( - __import__("os").environ.get( - "OPENCODE_SESSION_CONTINUATION_BUDGET", - str(DEFAULT_CONTINUATION_BUDGET), - ) - ), - ) + append.add_argument("--budget", required=True, type=int) budget = subparsers.add_parser( "budget-remaining", help="Print remaining same-model continuation budget." ) budget.add_argument("--checkpoint", required=True, type=Path) - budget.add_argument( - "--budget", - type=int, - default=int( - __import__("os").environ.get( - "OPENCODE_SESSION_CONTINUATION_BUDGET", - str(DEFAULT_CONTINUATION_BUDGET), - ) - ), - ) + budget.add_argument("--budget", required=True, type=int) return parser.parse_args(argv) From 4070c161685298a37554a551e7e3f33b2b6e49ad Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:02:28 +0900 Subject: [PATCH 16/43] fix(opencode): fail closed without calibrated continuation budget --- scripts/ci/run_opencode_review_model_pool.sh | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/scripts/ci/run_opencode_review_model_pool.sh b/scripts/ci/run_opencode_review_model_pool.sh index 45969a30d5..117d0ea301 100644 --- a/scripts/ci/run_opencode_review_model_pool.sh +++ b/scripts/ci/run_opencode_review_model_pool.sh @@ -218,9 +218,13 @@ record_session_checkpoint() { append_same_model_continuation() { local prompt_file="$1" local checkpoint_file="$2" - local budget="${OPENCODE_SESSION_CONTINUATION_BUDGET:-2}" + local budget="${OPENCODE_SESSION_CONTINUATION_BUDGET:-}" local remaining + if ! is_non_negative_integer "$budget"; then + printf 'OpenCode continuation budget authority is not configured; checkpoint appendix injection fails closed.\n' + return 1 + fi remaining="$(python3 "$GITHUB_WORKSPACE/scripts/ci/opencode_review_session_checkpoint.py" append-continuation \ --prompt "$prompt_file" \ --checkpoint "$checkpoint_file" \ From 16e44666bb2f2a80ca51f3578c44efe4d0f3a4c7 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:03:47 +0900 Subject: [PATCH 17/43] docs(opencode): record budget authority fail-closed boundary --- CHANGELOG.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 942631cb3c..be1f13ac02 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -96,7 +96,7 @@ - Raised `hourly-review-repair.yml`'s discovery ceiling from 50 to 200 while rotating deterministic 50-PR deep-inspection windows by hourly run number. The scheduler hydrates only the selected window and stops immediately after its single dispatch, preserving access to newer PRs without quadrupling expensive review/check/comment work. See `docs/doctoring/hourly-review-repair-single-file-consolidation.md`'s 2026-09-03 follow-up. ## [Unreleased] -- **Keep OpenCode same-model checkpoint evidence bounded and route-scoped.** PR #2284 now reads only declared log prefixes, counts required markers only from assistant text parts, rebuilds the base prompt before each retry so checkpoint appendices do not accumulate, and creates/consumes continuation state only for `contextual-orchestrator/orchestrator/free`. Four test-first regressions cover the prior behavior; doctoring records exact RED→GREEN evidence. +- **Keep OpenCode same-model checkpoint evidence bounded, route-scoped, and authority-driven.** PR #2284 reads only declared log prefixes, counts required markers only from assistant text parts, rebuilds the base prompt before each retry, and creates/consumes continuation state only for `contextual-orchestrator/orchestrator/free`. Missing calibrated budget authority now fails closed without an appendix instead of silently selecting two continuations; explicit budgets remain Proposed until A/B and allocator evidence exist. Six test-first regressions cover the repaired behavior. - **Bind GitHub REST redirect evidence to both production opener chains.** `.github#2279` now feeds a synthetic same-authority 302 through the CodeQL identity and Strix evidence clients' real module-level openers, proving the redirect target is never contacted and the bearer header is never forwarded. Removing `_RejectRedirects` from either opener makes the contract fail on the forbidden second request. Four stale Strix HTTP/transport/JSON fixtures now patch that same production seam; direct handler unit cases and standalone CodeQL materialization remain unchanged. - **Define an evidence-backed repository README quality standard.** Added `docs/repository-readme-quality-standard.md` as the shared review contract for product-first structure, code-current onboarding, authority boundaries, durable quality signals, and repository/source/dependency license due diligence. Product repositories continue to own their own README prose; the standard is linked from the root documentation map and does not centralize or generate product claims. - Include merge-scheduler entrypoint, core, and regression-test changes in From a197670b3ac0fcfeaada50c5aae3a65fdc90472c Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:04:17 +0900 Subject: [PATCH 18/43] docs(opencode): doctor continuation budget authority RCA --- .../opencode-same-model-midabort-20260919.md | 33 +++++++++++++++++-- 1 file changed, 30 insertions(+), 3 deletions(-) diff --git a/docs/doctoring/opencode-same-model-midabort-20260919.md b/docs/doctoring/opencode-same-model-midabort-20260919.md index 6e5cf574ab..936baf5401 100644 --- a/docs/doctoring/opencode-same-model-midabort-20260919.md +++ b/docs/doctoring/opencode-same-model-midabort-20260919.md @@ -27,8 +27,8 @@ Pinned model for measurement: `contextual-orchestrator/orchestrator/free` - `scripts/ci/opencode_review_session_checkpoint.py` — host-managed checkpoint ledger (digest-only partial work, missing required outputs, route telemetry). - `scripts/ci/run_opencode_review_model_pool.sh` — injects bounded same-model - continuation appendix on retry (`OPENCODE_SESSION_CONTINUATION_BUDGET`, default - `2`). + continuation appendix on retry only when explicit calibrated authority supplies + `OPENCODE_SESSION_CONTINUATION_BUDGET`; missing authority injects no appendix. ## Before / after (pinned model, fixture-backed) @@ -37,7 +37,7 @@ Pinned model for measurement: `contextual-orchestrator/orchestrator/free` | Same-model retry carries prior termination reason | No | Yes (checkpoint) | | Same-model retry carries missing required outputs | No | Yes | | Route attempt telemetry on retry | Discarded | Preserved (CO#1205 allowlist) | -| Continuation budget | Unbounded prompt replay cycles only | Explicit env budget, tested | +| Continuation budget | Unbounded prompt replay cycles only | Missing authority fails closed; explicit budget path is tested but remains Proposed pending calibration | | False success on incomplete control | Fail-closed already | Unchanged fail-closed | | Partial provider body replayed into prompt | N/A | Forbidden (digest only) | @@ -75,3 +75,30 @@ warnings-as-errors execution passed **43 tests** across the checkpoint, route evidence, and model-pool suites; Bash syntax, Python compilation, and diff whitespace checks also passed. Hosted exact-head gates remain separately required and no predecessor receipt transfers. + + +### Continuation budget authority repair (2026-09-20) + +Review of exact `bed37694c191f13bf18a595af672bb4d63e811af` found that +`DEFAULT_CONTINUATION_BUDGET = 2` and +`${OPENCODE_SESSION_CONTINUATION_BUDGET:-2}` silently selected two +decision-affecting continuations while the controlled A/B and allocator +evidence remained open. This violated the no-heuristics contract. + +Test-only commits `f0775fd48d31bf033279701f747698310aa6d7c6` and +`c4165352bab48123f9333bcfa11a78d060cb9cb0` require the runner and direct +CLI to reject missing budget authority. Minimal source commits +`3e290447b80c3964599bbb811f5cada2435ce28f` and +`4070c161685298a37554a551e7e3f33b2b6e49ad` remove both numeric defaults: +the runner injects no checkpoint appendix without an explicit non-negative +budget, and both CLI commands require `--budget`. + +Fresh materialization of exact `4070c161…` passed Python compilation, full +runner Bash syntax, and four direct authority probes (missing CLI authority +fails, the required argument is diagnosed, an explicit budget remains +accepted, and the shell numeric fallback is absent). The execution image does +not contain pytest, so no fresh pytest count is claimed. An explicit budget is +still Proposed rather than calibrated production authority until controlled +completion/time/token evidence and the selected fast-mlsirm/Fugu/Conductor/ +TRINITY-compatible allocator receipt are integrated. Hosted exact-head gates +and independent review remain required. From c239d22141dac9b3a6f3eb536577e5a53a6d3020 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:04:35 +0900 Subject: [PATCH 19/43] docs(gap): track continuation budget authority --- docs/product-technical-gap-baseline.md | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/product-technical-gap-baseline.md b/docs/product-technical-gap-baseline.md index a713afccda..a78787a85d 100644 --- a/docs/product-technical-gap-baseline.md +++ b/docs/product-technical-gap-baseline.md @@ -12,6 +12,7 @@ | Gap ID | 상태 | exact-head evidence | causal owner / next gate | |---|---|---|---| | CONTROL-OPENCODE-CHECKPOINT-INTEGRITY-01 | **Source repaired on PR #2284; hosted exact-head acceptance pending** | Review of `#2284@9fdfddfa` found complete-file reads before slicing, user-prompt marker laundering, accumulated retry appendices, and checkpoint application outside `contextual-orchestrator/orchestrator/free`. Test-only `afe1420d` produced exactly 4 failures; source `b400ad5d` produced 43 focused warnings-as-errors passes. Later test expansion at `9012eac2` hid an arithmetically unreachable `used < 0` decision from coverage while its named test exercised only `used == 0`. RED `802a4fa5` makes that vacuous oracle executable; GREEN `0aa9b902` removes only the impossible clamp. Exact-source compile and budget probes pass for empty through exhausted histories. | ContextualWisdomLab/.github owns the trusted OpenCode host checkpoint boundary. Keep #2284 Draft until its documentation successor receives fresh terminal hosted security/quality evidence and qualifying independent review; no predecessor result transfers. | +| CONTROL-OPENCODE-CONTINUATION-AUTHORITY-02 | **Missing authority now fails closed on PR #2284; calibrated admission remains Proposed** | Exact `bed37694` silently selected budget `2` although controlled completion/time/token evidence was still pending. RED `f0775fd4` and `c4165352` require the runner and direct CLI to reject absent authority. GREEN `3e290447` and `4070c161` remove the Python and shell defaults. Fresh exact materialization passed compile, Bash syntax, and direct authority probes **4/4**; pytest is unavailable in the execution image and is not claimed. | ContextualWisdomLab/.github owns host enforcement. A budget may be enabled only after a versioned fast-mlsirm/Fugu/Conductor/TRINITY-compatible allocator receipt and controlled A/B evidence are integrated; absent authority keeps checkpoint injection disabled. Hosted exact-head GREEN and independent review remain required. | ### 2026-09-13 current-head incident delta From 7c5c6a7548f5836d956cb196427581060951021b Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:06:14 +0900 Subject: [PATCH 20/43] test(opencode): exclude provider identity from continuation prompts --- tests/test_opencode_review_session_checkpoint.py | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/tests/test_opencode_review_session_checkpoint.py b/tests/test_opencode_review_session_checkpoint.py index b4c7962725..e93615173c 100644 --- a/tests/test_opencode_review_session_checkpoint.py +++ b/tests/test_opencode_review_session_checkpoint.py @@ -78,8 +78,8 @@ def test_classify_termination_detects_provider_fatal(tmp_path: Path) -> None: assert reason == "provider-fatal" -def test_record_and_continue_preserves_route_evidence(tmp_path: Path) -> None: - """Same-model retries carry route telemetry without replaying provider bodies.""" +def test_record_and_continue_excludes_provider_identity_from_prompt(tmp_path: Path) -> None: + """Leaf continuation prompts cannot consume provider-specific route telemetry.""" export_path = tmp_path / "export.json" export_path.write_text(_export_with_text("in progress"), encoding="utf-8") route_path = tmp_path / "route.json" @@ -120,8 +120,10 @@ def test_record_and_continue_preserves_route_evidence(tmp_path: Path) -> None: appendix = build_continuation_appendix(checkpoint_path, budget=2) assert "contextual-orchestrator/orchestrator/free" in appendix assert "termination reason" in appendix.casefold() - assert "route evidence" in appendix.casefold() - assert "provider_attempt_count=1" in appendix + assert "route evidence" not in appendix.casefold() + assert "openrouter" not in appendix.casefold() + assert "connecting" not in appendix.casefold() + assert "429" not in appendix assert "in progress" not in appendix From 3a8c056bc2b4e0f5b012f900b24bfc54636e24ba Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:06:48 +0900 Subject: [PATCH 21/43] test(opencode): bind provider-neutral continuation contract --- ...test_opencode_review_session_checkpoint.py | 32 ++++++------------- 1 file changed, 9 insertions(+), 23 deletions(-) diff --git a/tests/test_opencode_review_session_checkpoint.py b/tests/test_opencode_review_session_checkpoint.py index e93615173c..f460980136 100644 --- a/tests/test_opencode_review_session_checkpoint.py +++ b/tests/test_opencode_review_session_checkpoint.py @@ -82,28 +82,6 @@ def test_record_and_continue_excludes_provider_identity_from_prompt(tmp_path: Pa """Leaf continuation prompts cannot consume provider-specific route telemetry.""" export_path = tmp_path / "export.json" export_path.write_text(_export_with_text("in progress"), encoding="utf-8") - route_path = tmp_path / "route.json" - route_path.write_text( - json.dumps( - { - "error": { - "detail": { - "model": "orchestrator/free", - "terminal_reason": "rate_limited", - "attempts": [ - { - "provider_name": "openrouter", - "phase": "connecting", - "attempt_number": 1, - "provider_status": 429, - } - ], - } - } - } - ), - encoding="utf-8", - ) checkpoint_path = tmp_path / "checkpoint.json" record_attempt_checkpoint( checkpoint_path=checkpoint_path, @@ -115,8 +93,16 @@ def test_record_and_continue_excludes_provider_identity_from_prompt(tmp_path: Pa json_path=tmp_path / "run.jsonl", export_path=export_path, exit_code=1, - route_evidence_path=route_path, ) + document = json.loads(checkpoint_path.read_text(encoding="utf-8")) + document["attempts"][-1]["route_telemetry"] = { + "provider_attempt_count": 1, + "provider_name": "openrouter", + "upstream_phase": "connecting", + "upstream_status": 429, + "served_model": "orchestrator/free", + } + checkpoint_path.write_text(json.dumps(document), encoding="utf-8") appendix = build_continuation_appendix(checkpoint_path, budget=2) assert "contextual-orchestrator/orchestrator/free" in appendix assert "termination reason" in appendix.casefold() From 866cc6a4a1285936da5110ea0a28071548da09e2 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:07:16 +0900 Subject: [PATCH 22/43] fix(opencode): keep provider identity out of continuation state --- .../ci/opencode_review_session_checkpoint.py | 25 +------------------ 1 file changed, 1 insertion(+), 24 deletions(-) diff --git a/scripts/ci/opencode_review_session_checkpoint.py b/scripts/ci/opencode_review_session_checkpoint.py index 2b34ffd0d4..a2b11bebba 100644 --- a/scripts/ci/opencode_review_session_checkpoint.py +++ b/scripts/ci/opencode_review_session_checkpoint.py @@ -21,12 +21,6 @@ if str(_REPO_ROOT) not in sys.path: # pragma: no cover - bootstrap only on direct script execution sys.path.insert(0, str(_REPO_ROOT)) -from scripts.ci.contextual_orchestrator_route_evidence import ( # noqa: E402 - extract_gateway_route_telemetry, - format_route_telemetry, -) - - CHECKPOINT_SCHEMA = 1 CONTROL_SENTINEL = "opencode-review-control-v1" REQUIRED_OUTPUT_MARKERS = ( @@ -175,7 +169,6 @@ def record_attempt_checkpoint( export_path: Path, exit_code: int, stderr_path: Path | None = None, - route_evidence_path: Path | None = None, log_hint: str = "", ) -> dict[str, Any]: """Append one bounded attempt record to the host checkpoint ledger.""" @@ -207,11 +200,6 @@ def record_attempt_checkpoint( partial_text = "\n".join(chunks) except (OSError, json.JSONDecodeError, UnicodeDecodeError): partial_text = "" - route_telemetry: dict[str, str | int] = {} - if route_evidence_path and route_evidence_path.is_file(): - route_telemetry = extract_gateway_route_telemetry( - _read_bounded_text(route_evidence_path, 65536) - ) entry = { "attempt": attempt, "model_candidate": model_candidate, @@ -228,7 +216,6 @@ def record_attempt_checkpoint( "exit_code": exit_code, "partial_summary": partial_summary, "missing_required_outputs": missing_required_outputs(partial_text), - "route_telemetry": route_telemetry, } document = _load_checkpoint(checkpoint_path) attempts = document.setdefault("attempts", []) @@ -236,8 +223,6 @@ def record_attempt_checkpoint( attempts.append(entry) document["pinned_model"] = model_candidate document["head_sha"] = head_sha - if route_telemetry: - document["last_route_telemetry"] = route_telemetry checkpoint_path.parent.mkdir(parents=True, exist_ok=True) checkpoint_path.write_text( json.dumps(document, indent=2, sort_keys=True) + "\n", @@ -288,9 +273,6 @@ def build_continuation_appendix(checkpoint_path: Path, *, budget: int) -> str: "- Required outputs still missing from the prior attempt: " + ", ".join(f"`{item}`" for item in missing[:8]) ) - route = last.get("route_telemetry") - if isinstance(route, Mapping) and route: - lines.append(f"- Route evidence: {format_route_telemetry(dict(route))}") lines.extend( [ "- Resume from the trusted evidence packet and complete every required output.", @@ -334,7 +316,6 @@ def parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: record.add_argument("--export-path", required=True, type=Path) record.add_argument("--exit-code", required=True, type=int) record.add_argument("--stderr-path", type=Path) - record.add_argument("--route-evidence-path", type=Path) record.add_argument("--log-hint", default="") append = subparsers.add_parser( @@ -367,15 +348,11 @@ def main(argv: Sequence[str] | None = None) -> int: export_path=args.export_path, exit_code=args.exit_code, stderr_path=args.stderr_path, - route_evidence_path=args.route_evidence_path, log_hint=args.log_hint, ) print( json.dumps( - { - "termination_reason": entry["termination_reason"], - "route_telemetry": format_route_telemetry(entry["route_telemetry"]), - }, + {"termination_reason": entry["termination_reason"]}, separators=(",", ":"), ) ) From 8c04d1284f96eb0a702a24287290e30ba3f1bb9d Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:07:27 +0900 Subject: [PATCH 23/43] fix(opencode): remove provider route evidence from leaf runner --- scripts/ci/run_opencode_review_model_pool.sh | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/scripts/ci/run_opencode_review_model_pool.sh b/scripts/ci/run_opencode_review_model_pool.sh index 117d0ea301..55f8bac3f7 100644 --- a/scripts/ci/run_opencode_review_model_pool.sh +++ b/scripts/ci/run_opencode_review_model_pool.sh @@ -198,8 +198,6 @@ record_session_checkpoint() { local export_file="$5" local exit_code="$6" local stderr_file="$7" - local route_evidence_file="${8:-}" - PYTHONPATH="$GITHUB_WORKSPACE${PYTHONPATH:+:$PYTHONPATH}" python3 "$GITHUB_WORKSPACE/scripts/ci/opencode_review_session_checkpoint.py" record \ --checkpoint "$checkpoint_file" \ --model-candidate "$model_candidate" \ @@ -211,7 +209,6 @@ record_session_checkpoint() { --export-path "$export_file" \ --exit-code "$exit_code" \ ${stderr_file:+--stderr-path "$stderr_file"} \ - ${route_evidence_file:+--route-evidence-path "$route_evidence_file"} \ || true } @@ -583,7 +580,6 @@ main() { opencode_json_file="${candidate_output_file}.jsonl" opencode_export_file="${candidate_output_file}.session.json" checkpoint_file="${RUNNER_TEMP}/opencode-checkpoint-${safe_model}.json" - route_evidence_file="${OPENCODE_ROUTE_EVIDENCE_FILE:-}" write_prompt "$model_candidate" "$prompt_file" effective_attempts="$attempts" if is_schema_repair_candidate "$model_candidate"; then @@ -633,7 +629,7 @@ main() { && [ "$run_status" -ne 0 ] && [ "$run_status" -ne 2 ]; then record_session_checkpoint "$model_candidate" "$attempt" "$checkpoint_file" \ "$opencode_json_file" "$opencode_export_file" "$run_status" \ - "${opencode_json_file}.stderr" "$route_evidence_file" + "${opencode_json_file}.stderr" fi if [ "$run_status" -ne 3 ] && is_credit_exhausted_failure "$opencode_json_file" "${opencode_json_file}.stderr"; then dead_candidate_reasons[$model_candidate]="provider credits exhausted (HTTP 402 / payment required)" From 51a531886962f35eb091b97cf8ec2d6e05dcec81 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:07:41 +0900 Subject: [PATCH 24/43] refactor(opencode): remove mutable CO schema copy --- .../contextual_orchestrator_route_evidence.py | 110 ------------------ 1 file changed, 110 deletions(-) delete mode 100644 scripts/ci/contextual_orchestrator_route_evidence.py diff --git a/scripts/ci/contextual_orchestrator_route_evidence.py b/scripts/ci/contextual_orchestrator_route_evidence.py deleted file mode 100644 index 8401342e1a..0000000000 --- a/scripts/ci/contextual_orchestrator_route_evidence.py +++ /dev/null @@ -1,110 +0,0 @@ -"""Consume bounded contextual-orchestrator route attempt telemetry. - -Typed ``attempts[]`` and ``terminal_reason`` on gateway error envelopes were -added for ``route_once`` failover in ContextualWisdomLab/contextual-orchestrator#1205 -(closes #1016). OpenCode and Noema consumers share this allowlisted parser so -same-model retries preserve provider-outcome evidence without re-emitting raw -provider bodies. -""" - -from __future__ import annotations - -import json -import re -from typing import Any - - -SAFE_MODEL_IDENTIFIER_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:/@+-]{0,199}$") -MAX_HTTP_ERROR_BODY_BYTES = 65536 - - -def safe_model_identifier(value: Any) -> str | None: - """Return a conservative model or route identifier safe for public logs.""" - if not isinstance(value, str): - return None - candidate = value.strip() - if not SAFE_MODEL_IDENTIFIER_RE.fullmatch(candidate): - return None - return candidate - - -def extract_gateway_route_telemetry(raw: str | bytes) -> dict[str, str | int]: - """Parse allowlisted route evidence from one gateway HTTP error envelope. - - Only ``error.detail`` scalar fields and bounded ``attempts[]`` entries are - retained. Malformed, oversized, or present-but-unsafe ``model`` / - ``terminal_reason`` values fail closed to an empty mapping so allowlisted - attempt fields cannot launder free-form error text. - """ - if isinstance(raw, bytes): - if len(raw) > MAX_HTTP_ERROR_BODY_BYTES: - return {} - try: - raw_text = raw.decode("utf-8") - except UnicodeDecodeError: - return {} - else: - raw_text = raw - try: - payload = json.loads(raw_text) - except (json.JSONDecodeError, TypeError, ValueError): - return {} - if not isinstance(payload, dict): - return {} - error = payload.get("error") - if not isinstance(error, dict): - return {} - detail = error.get("detail") - if not isinstance(detail, dict): - return {} - raw_model = detail.get("model") - if raw_model is not None: - model = safe_model_identifier(raw_model) - if model is None: - return {} - else: - model = None - raw_terminal_reason = detail.get("terminal_reason") - if raw_terminal_reason is not None: - terminal_reason = safe_model_identifier(raw_terminal_reason) - if terminal_reason is None: - return {} - else: - terminal_reason = None - telemetry: dict[str, str | int] = {} - attempts = detail.get("attempts") - if model is not None: - telemetry["served_model"] = model - if terminal_reason is not None: - telemetry["terminal_reason"] = terminal_reason - if isinstance(attempts, list) and attempts and len(attempts) <= 64: - telemetry["provider_attempt_count"] = len(attempts) - last_attempt = attempts[-1] - if isinstance(last_attempt, dict): - provider_name = safe_model_identifier(last_attempt.get("provider_name")) - phase = safe_model_identifier(last_attempt.get("phase")) - attempt_number = last_attempt.get("attempt_number") - provider_status = last_attempt.get("provider_status") - if provider_name is not None: - telemetry["provider_name"] = provider_name - if phase is not None: - telemetry["upstream_phase"] = phase - if type(attempt_number) is int and 1 <= attempt_number <= 64: - telemetry["attempt_number"] = attempt_number - if type(provider_status) is int and 100 <= provider_status <= 599: - telemetry["upstream_status"] = provider_status - return telemetry - - -def format_route_telemetry(telemetry: dict[str, str | int]) -> str: - """Format allowlisted route telemetry for bounded Actions logs.""" - ordered_keys = ( - "provider_attempt_count", - "provider_name", - "upstream_phase", - "attempt_number", - "upstream_status", - "terminal_reason", - "served_model", - ) - return " ".join(f"{key}={telemetry[key]}" for key in ordered_keys if key in telemetry) From b7e8256f4c8a9cf7cf1022312ee86ba6f985f9a4 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:07:46 +0900 Subject: [PATCH 25/43] test(opencode): retire consumer-owned route parser fixtures --- ..._contextual_orchestrator_route_evidence.py | 181 ------------------ 1 file changed, 181 deletions(-) delete mode 100644 tests/test_contextual_orchestrator_route_evidence.py diff --git a/tests/test_contextual_orchestrator_route_evidence.py b/tests/test_contextual_orchestrator_route_evidence.py deleted file mode 100644 index fe6271b7b2..0000000000 --- a/tests/test_contextual_orchestrator_route_evidence.py +++ /dev/null @@ -1,181 +0,0 @@ -"""Tests for bounded contextual-orchestrator route attempt telemetry.""" - -from __future__ import annotations - -import json - -from scripts.ci.contextual_orchestrator_route_evidence import ( - extract_gateway_route_telemetry, - format_route_telemetry, -) - - -def test_extract_gateway_route_telemetry_reads_typed_attempts() -> None: - """CO#1205 typed attempts[] fields are preserved without raw provider bodies.""" - payload = { - "error": { - "detail": { - "model": "openrouter/meta-llama/llama-3.3-70b-instruct:free", - "terminal_reason": "eligible_candidates_exhausted", - "attempts": [ - { - "provider_name": "openrouter", - "phase": "connecting", - "attempt_number": 1, - "provider_status": 429, - }, - { - "provider_name": "nvidia_nim", - "phase": "streaming", - "attempt_number": 2, - "provider_status": 502, - }, - ], - } - } - } - telemetry = extract_gateway_route_telemetry(json.dumps(payload)) - assert telemetry["served_model"] == "openrouter/meta-llama/llama-3.3-70b-instruct:free" - assert telemetry["terminal_reason"] == "eligible_candidates_exhausted" - assert telemetry["provider_attempt_count"] == 2 - assert telemetry["provider_name"] == "nvidia_nim" - assert telemetry["upstream_phase"] == "streaming" - assert telemetry["attempt_number"] == 2 - assert telemetry["upstream_status"] == 502 - - -def test_extract_gateway_route_telemetry_rejects_unsafe_identifiers() -> None: - """Free-form provider error text must not become public telemetry.""" - payload = { - "error": { - "detail": { - "model": "bad\nmodel", - "terminal_reason": "x" * 300, - "attempts": [{"provider_name": "ok", "attempt_number": 999}], - } - } - } - # Present-but-unsafe model/terminal_reason fail closed for the whole - # envelope so allowlisted attempt fields cannot launder free-form text. - assert extract_gateway_route_telemetry(json.dumps(payload)) == {} - - -def test_format_route_telemetry_is_stable() -> None: - """Formatted telemetry uses a fixed key order for log correlation.""" - formatted = format_route_telemetry( - { - "terminal_reason": "pool_exhausted", - "provider_attempt_count": 3, - "served_model": "orchestrator/free", - } - ) - assert formatted == ( - "provider_attempt_count=3 terminal_reason=pool_exhausted served_model=orchestrator/free" - ) - - -def test_extract_gateway_route_telemetry_fail_closed_cases() -> None: - """Malformed or oversized envelopes never emit partial unsafe telemetry.""" - from scripts.ci.contextual_orchestrator_route_evidence import ( - MAX_HTTP_ERROR_BODY_BYTES, - safe_model_identifier, - ) - - assert safe_model_identifier(42) is None - assert extract_gateway_route_telemetry(b"x" * (MAX_HTTP_ERROR_BODY_BYTES + 1)) == {} - assert extract_gateway_route_telemetry(b"\xff\xfe") == {} - assert extract_gateway_route_telemetry("not-json") == {} - assert extract_gateway_route_telemetry("[]") == {} - assert extract_gateway_route_telemetry(json.dumps({"error": "x"})) == {} - assert extract_gateway_route_telemetry(json.dumps({"error": {"detail": "x"}})) == {} - assert ( - extract_gateway_route_telemetry( - json.dumps({"error": {"detail": {"model": "bad\nmodel", "attempts": []}}}) - ) - == {} - ) - assert ( - extract_gateway_route_telemetry( - json.dumps( - { - "error": { - "detail": { - "terminal_reason": "x" * 300, - "attempts": [{"provider_name": "ok"}], - } - } - } - ) - ) - == {} - ) - - -def test_extract_gateway_route_telemetry_ignores_empty_or_oversized_attempt_lists() -> None: - """Empty or oversized attempt lists do not emit attempt telemetry.""" - assert ( - extract_gateway_route_telemetry(json.dumps({"error": {"detail": {"attempts": []}}})) - == {} - ) - attempts = [{"provider_name": "openrouter", "attempt_number": 1}] * 65 - assert ( - extract_gateway_route_telemetry(json.dumps({"error": {"detail": {"attempts": attempts}}})) - == {} - ) - assert ( - extract_gateway_route_telemetry(json.dumps({"error": {"detail": {"attempts": "bad"}}})) - == {} - ) - - -def test_extract_gateway_route_telemetry_ignores_non_object_attempt_rows() -> None: - """Attempt rows must be objects before any subfield is read.""" - payload = {"error": {"detail": {"attempts": ["not-an-object"]}}} - assert extract_gateway_route_telemetry(json.dumps(payload)) == { - "provider_attempt_count": 1 - } - - -def test_extract_gateway_route_telemetry_ignores_invalid_attempt_fields() -> None: - """Invalid attempt subfields are omitted without rejecting the envelope.""" - payload = { - "error": { - "detail": { - "model": "orchestrator/free", - "attempts": [ - { - "provider_name": "bad name", - "phase": "bad phase", - "attempt_number": 999, - "provider_status": 999, - } - ], - } - } - } - telemetry = extract_gateway_route_telemetry(json.dumps(payload)) - assert telemetry == {"served_model": "orchestrator/free", "provider_attempt_count": 1} - - -def test_extract_gateway_route_telemetry_optional_attempt_fields() -> None: - """Allowlisted attempt fields are optional and independently validated.""" - payload = { - "error": { - "detail": { - "attempts": [ - { - "provider_name": "openrouter", - "phase": "streaming", - "attempt_number": 2, - "provider_status": 503, - } - ] - } - } - } - telemetry = extract_gateway_route_telemetry(json.dumps(payload)) - assert telemetry["provider_attempt_count"] == 1 - assert telemetry["provider_name"] == "openrouter" - assert telemetry["upstream_phase"] == "streaming" - assert telemetry["attempt_number"] == 2 - assert telemetry["upstream_status"] == 503 From e76de7c0e1bcd7e733b33db21fdf24a27ba651da Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:09:01 +0900 Subject: [PATCH 26/43] docs(opencode): record provider-neutral continuation boundary --- CHANGELOG.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index be1f13ac02..6623b5d234 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -96,7 +96,7 @@ - Raised `hourly-review-repair.yml`'s discovery ceiling from 50 to 200 while rotating deterministic 50-PR deep-inspection windows by hourly run number. The scheduler hydrates only the selected window and stops immediately after its single dispatch, preserving access to newer PRs without quadrupling expensive review/check/comment work. See `docs/doctoring/hourly-review-repair-single-file-consolidation.md`'s 2026-09-03 follow-up. ## [Unreleased] -- **Keep OpenCode same-model checkpoint evidence bounded, route-scoped, and authority-driven.** PR #2284 reads only declared log prefixes, counts required markers only from assistant text parts, rebuilds the base prompt before each retry, and creates/consumes continuation state only for `contextual-orchestrator/orchestrator/free`. Missing calibrated budget authority now fails closed without an appendix instead of silently selecting two continuations; explicit budgets remain Proposed until A/B and allocator evidence exist. Six test-first regressions cover the repaired behavior. +- **Keep OpenCode same-model checkpoint evidence bounded, provider-neutral, and authority-driven.** PR #2284 reads only declared log prefixes, counts required markers only from assistant text parts, rebuilds the base prompt before each retry, and creates/consumes continuation state only for `contextual-orchestrator/orchestrator/free`. Missing calibrated budget authority injects no appendix. Provider/model/status route details remain inside CO and cannot enter the leaf checkpoint or prompt; the mutable consumer-side CO parser and fixtures were removed. Explicit budgets remain Proposed until A/B, fast-mlsirm, and allocator evidence exist. - **Bind GitHub REST redirect evidence to both production opener chains.** `.github#2279` now feeds a synthetic same-authority 302 through the CodeQL identity and Strix evidence clients' real module-level openers, proving the redirect target is never contacted and the bearer header is never forwarded. Removing `_RejectRedirects` from either opener makes the contract fail on the forbidden second request. Four stale Strix HTTP/transport/JSON fixtures now patch that same production seam; direct handler unit cases and standalone CodeQL materialization remain unchanged. - **Define an evidence-backed repository README quality standard.** Added `docs/repository-readme-quality-standard.md` as the shared review contract for product-first structure, code-current onboarding, authority boundaries, durable quality signals, and repository/source/dependency license due diligence. Product repositories continue to own their own README prose; the standard is linked from the root documentation map and does not centralize or generate product claims. - Include merge-scheduler entrypoint, core, and regression-test changes in From d4d0ab022bae112724f0d4541948f9c596229cfd Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:09:21 +0900 Subject: [PATCH 27/43] docs(opencode): doctor provider-neutral owner boundary --- .../opencode-same-model-midabort-20260919.md | 37 ++++++++++++++++--- 1 file changed, 32 insertions(+), 5 deletions(-) diff --git a/docs/doctoring/opencode-same-model-midabort-20260919.md b/docs/doctoring/opencode-same-model-midabort-20260919.md index 936baf5401..d8f9c4d87c 100644 --- a/docs/doctoring/opencode-same-model-midabort-20260919.md +++ b/docs/doctoring/opencode-same-model-midabort-20260919.md @@ -20,10 +20,10 @@ Pinned model for measurement: `contextual-orchestrator/orchestrator/free` ## Reused surfaces -- `scripts/ci/contextual_orchestrator_route_evidence.py` — consumer for typed - `attempts[]` / `terminal_reason` envelopes from - `ContextualWisdomLab/contextual-orchestrator#1205` (closes #1016); mirrors the - allowlist already used in `noema_review_gate.py`. +- Provider-specific route evidence remains inside ContextualWisdomLab/contextual-orchestrator. + The leaf does not consume mutable `attempts[]`, provider, model, phase, or status + fields from open CO PR #1205; a released provider-neutral projection is required + before any owner telemetry can enter continuation control context. - `scripts/ci/opencode_review_session_checkpoint.py` — host-managed checkpoint ledger (digest-only partial work, missing required outputs, route telemetry). - `scripts/ci/run_opencode_review_model_pool.sh` — injects bounded same-model @@ -36,7 +36,7 @@ Pinned model for measurement: `contextual-orchestrator/orchestrator/free` | --- | --- | --- | | Same-model retry carries prior termination reason | No | Yes (checkpoint) | | Same-model retry carries missing required outputs | No | Yes | -| Route attempt telemetry on retry | Discarded | Preserved (CO#1205 allowlist) | +| Provider-specific route telemetry on retry | Discarded | Still discarded at the leaf; CO remains the observability owner | | Continuation budget | Unbounded prompt replay cycles only | Missing authority fails closed; explicit budget path is tested but remains Proposed pending calibration | | False success on incomplete control | Fail-closed already | Unchanged fail-closed | | Partial provider body replayed into prompt | N/A | Forbidden (digest only) | @@ -102,3 +102,30 @@ still Proposed rather than calibrated production authority until controlled completion/time/token evidence and the selected fast-mlsirm/Fugu/Conductor/ TRINITY-compatible allocator receipt are integrated. Hosted exact-head gates and independent review remain required. + + +### Provider-neutral owner-boundary repair (2026-09-20) + +The existing P0 review showed that the leaf parser read the mutable CO #1205 +`model`, `provider_name`, phase, attempt number, and HTTP status fields and +formatted them into the next model prompt. That made two envelopes with the +same provider-neutral outcome produce different control context and duplicated +an unreleased owner schema inside ContextualWisdomLab/.github. + +RED `7c5c6a7548f5836d956cb196427581060951021b` and +`3a8c056bc2b4e0f5b012f900b24bfc54636e24ba` require provider/model/phase/ +status values to have zero effect on the appendix. GREEN +`866cc6a4a1285936da5110ea0a28071548da09e2` and +`8c04d1284f96eb0a702a24287290e30ba3f1bb9d` remove route parsing, storage, +formatting, CLI plumbing, and runner input. Commits +`51a531886962f35eb091b97cf8ec2d6e05dcec81` and +`b7e8256f4c8a9cf7cf1022312ee86ba6f985f9a4` retire the consumer-owned parser +and fixtures entirely. + +Fresh exact materialization at `b7e8256f…` passed Python compilation, Bash +syntax, and **11/11** direct authority/provider-neutral probes. The probe injects +hostile legacy `openrouter`, phase, HTTP 429, and served-model fields into a +checkpoint and proves none reaches the continuation appendix. The local image +still lacks pytest, so no fresh pytest count is claimed. CO issue #1106 records +the required immutable, provider-neutral, allocator/fast-mlsirm receipt before +any owner telemetry or budget can be consumed. From 407aa107fc65228261b5429fc3fb2eb9ac618404 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:09:32 +0900 Subject: [PATCH 28/43] docs(gap): track provider-neutral continuation owner --- docs/product-technical-gap-baseline.md | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/product-technical-gap-baseline.md b/docs/product-technical-gap-baseline.md index a78787a85d..a99bd74886 100644 --- a/docs/product-technical-gap-baseline.md +++ b/docs/product-technical-gap-baseline.md @@ -13,6 +13,7 @@ |---|---|---|---| | CONTROL-OPENCODE-CHECKPOINT-INTEGRITY-01 | **Source repaired on PR #2284; hosted exact-head acceptance pending** | Review of `#2284@9fdfddfa` found complete-file reads before slicing, user-prompt marker laundering, accumulated retry appendices, and checkpoint application outside `contextual-orchestrator/orchestrator/free`. Test-only `afe1420d` produced exactly 4 failures; source `b400ad5d` produced 43 focused warnings-as-errors passes. Later test expansion at `9012eac2` hid an arithmetically unreachable `used < 0` decision from coverage while its named test exercised only `used == 0`. RED `802a4fa5` makes that vacuous oracle executable; GREEN `0aa9b902` removes only the impossible clamp. Exact-source compile and budget probes pass for empty through exhausted histories. | ContextualWisdomLab/.github owns the trusted OpenCode host checkpoint boundary. Keep #2284 Draft until its documentation successor receives fresh terminal hosted security/quality evidence and qualifying independent review; no predecessor result transfers. | | CONTROL-OPENCODE-CONTINUATION-AUTHORITY-02 | **Missing authority now fails closed on PR #2284; calibrated admission remains Proposed** | Exact `bed37694` silently selected budget `2` although controlled completion/time/token evidence was still pending. RED `f0775fd4` and `c4165352` require the runner and direct CLI to reject absent authority. GREEN `3e290447` and `4070c161` remove the Python and shell defaults. Fresh exact materialization passed compile, Bash syntax, and direct authority probes **4/4**; pytest is unavailable in the execution image and is not claimed. | ContextualWisdomLab/.github owns host enforcement. A budget may be enabled only after a versioned fast-mlsirm/Fugu/Conductor/TRINITY-compatible allocator receipt and controlled A/B evidence are integrated; absent authority keeps checkpoint injection disabled. Hosted exact-head GREEN and independent review remain required. | +| CONTROL-OPENCODE-PROVIDER-NEUTRAL-03 | **Consumer schema copy removed on PR #2284; released CO projection pending** | Exact `bed37694` parsed CO model/provider/phase/status fields and formatted them into continuation prompts. RED `7c5c6a75`/`3a8c056b` requires byte-neutral handling of hostile provider details. GREEN `866cc6a4`/`8c04d128` removes route parsing and runner plumbing; `51a53188`/`b7e8256f` deletes the mutable parser and fixtures. Fresh exact compile, Bash syntax, and direct authority/provider-neutral probes are **11/11**; pytest is unavailable and not claimed. | ContextualWisdomLab/contextual-orchestrator issue #1106 owns the released provider-neutral allocation receipt. ContextualWisdomLab/.github must know only `orchestrator/free` and the gateway token; provider identities remain CO observability data. Keep Draft until immutable owner release/pin, exact-head GREEN, and independent review. | ### 2026-09-13 current-head incident delta From 684598172a82a6fc0cb24662d3ad425560ac6aa3 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:50:41 +0900 Subject: [PATCH 29/43] test(opencode): reject negative continuation authority --- ...test_opencode_review_session_checkpoint.py | 26 +++++++++++++++++++ 1 file changed, 26 insertions(+) diff --git a/tests/test_opencode_review_session_checkpoint.py b/tests/test_opencode_review_session_checkpoint.py index f460980136..9975163c5e 100644 --- a/tests/test_opencode_review_session_checkpoint.py +++ b/tests/test_opencode_review_session_checkpoint.py @@ -470,6 +470,32 @@ def test_budget_cli_requires_explicit_authority(tmp_path: Path) -> None: assert "--budget" in completed.stderr +def test_budget_cli_rejects_negative_authority(tmp_path: Path) -> None: + """Every CLI path rejects negative continuation-budget authority.""" + for command_arguments in ( + ["budget-remaining"], + ["append-continuation", "--prompt", "prompt.md"], + ): + completed = subprocess.run( + [ + sys.executable, + "scripts/ci/opencode_review_session_checkpoint.py", + *command_arguments, + "--checkpoint", + str(tmp_path / "checkpoint.json"), + "--budget", + "-1", + ], + cwd=Path(__file__).resolve().parents[1], + capture_output=True, + text=True, + check=False, + ) + + assert completed.returncode != 0 + assert "non-negative integer" in completed.stderr + + def test_main_entrypoint(tmp_path: Path) -> None: """Module entrypoint delegates to main().""" export_path = tmp_path / "export.json" From 7c5844ad2c5297e5ef7a45fef5c2978e8f1b1a53 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:50:52 +0900 Subject: [PATCH 30/43] fix(opencode): validate continuation budget authority --- .../ci/opencode_review_session_checkpoint.py | 90 +++++++++++-------- 1 file changed, 51 insertions(+), 39 deletions(-) diff --git a/scripts/ci/opencode_review_session_checkpoint.py b/scripts/ci/opencode_review_session_checkpoint.py index a2b11bebba..ba26a4cd4f 100644 --- a/scripts/ci/opencode_review_session_checkpoint.py +++ b/scripts/ci/opencode_review_session_checkpoint.py @@ -235,8 +235,10 @@ def continuation_budget_remaining(checkpoint_path: Path, *, budget: int) -> int: """Return how many same-model continuations remain for this checkpoint.""" document = _load_checkpoint(checkpoint_path) attempts = document.get("attempts") - used = len(attempts) - 1 if isinstance(attempts, list) and attempts else 0 - return max(0, budget - used) + used_continuations = ( + len(attempts) - 1 if isinstance(attempts, list) and attempts else 0 + ) + return max(0, budget - used_continuations) def build_continuation_appendix(checkpoint_path: Path, *, budget: int) -> str: @@ -245,33 +247,33 @@ def build_continuation_appendix(checkpoint_path: Path, *, budget: int) -> str: attempts = document.get("attempts") if not isinstance(attempts, list) or not attempts: return "" - remaining = continuation_budget_remaining(checkpoint_path, budget=budget) - if remaining <= 0: + remaining_budget = continuation_budget_remaining(checkpoint_path, budget=budget) + if remaining_budget <= 0: return "" - last = attempts[-1] - if not isinstance(last, Mapping): + latest_attempt = attempts[-1] + if not isinstance(latest_attempt, Mapping): return "" lines = [ "", "## Same-model continuation (host checkpoint; not approval evidence)", f"- Pinned model: `{document.get('pinned_model', 'unknown')}`", - f"- Prior attempt: `{last.get('attempt', '?')}`", - f"- Termination reason: `{last.get('termination_reason', 'unknown')}`", - f"- Continuation budget remaining after this attempt: `{remaining}`", + f"- Prior attempt: `{latest_attempt.get('attempt', '?')}`", + f"- Termination reason: `{latest_attempt.get('termination_reason', 'unknown')}`", + f"- Continuation budget remaining after this attempt: `{remaining_budget}`", ] - partial = last.get("partial_summary") - if isinstance(partial, Mapping): + partial_summary = latest_attempt.get("partial_summary") + if isinstance(partial_summary, Mapping): lines.append( "- Partial assistant digest: " - f"lines=`{partial.get('assistant_line_count', 0)}` " - f"sha256=`{partial.get('assistant_sha256', '')}` " - f"control_sentinel=`{partial.get('has_control_sentinel', False)}`" + f"lines=`{partial_summary.get('assistant_line_count', 0)}` " + f"sha256=`{partial_summary.get('assistant_sha256', '')}` " + f"control_sentinel=`{partial_summary.get('has_control_sentinel', False)}`" ) - missing = last.get("missing_required_outputs") - if isinstance(missing, list) and missing: + missing_outputs = latest_attempt.get("missing_required_outputs") + if isinstance(missing_outputs, list) and missing_outputs: lines.append( "- Required outputs still missing from the prior attempt: " - + ", ".join(f"`{item}`" for item in missing[:8]) + + ", ".join(f"`{item}`" for item in missing_outputs[:8]) ) lines.extend( [ @@ -300,36 +302,46 @@ def append_continuation_to_prompt( return continuation_budget_remaining(checkpoint_path, budget=budget) +def non_negative_integer(raw_value: str) -> int: + """Parse a non-negative integer for an explicit CLI authority.""" + parsed_value = int(raw_value) + if parsed_value < 0: + raise argparse.ArgumentTypeError("must be a non-negative integer") + return parsed_value + + def parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: """Parse checkpoint CLI commands.""" parser = argparse.ArgumentParser(description=__doc__) subparsers = parser.add_subparsers(dest="command", required=True) - record = subparsers.add_parser("record", help="Record one failed attempt checkpoint.") - record.add_argument("--checkpoint", required=True, type=Path) - record.add_argument("--model-candidate", required=True) - record.add_argument("--attempt", required=True, type=int) - record.add_argument("--head-sha", required=True) - record.add_argument("--run-id", required=True) - record.add_argument("--run-attempt", required=True) - record.add_argument("--json-path", required=True, type=Path) - record.add_argument("--export-path", required=True, type=Path) - record.add_argument("--exit-code", required=True, type=int) - record.add_argument("--stderr-path", type=Path) - record.add_argument("--log-hint", default="") - - append = subparsers.add_parser( + record_parser = subparsers.add_parser( + "record", help="Record one failed attempt checkpoint." + ) + record_parser.add_argument("--checkpoint", required=True, type=Path) + record_parser.add_argument("--model-candidate", required=True) + record_parser.add_argument("--attempt", required=True, type=int) + record_parser.add_argument("--head-sha", required=True) + record_parser.add_argument("--run-id", required=True) + record_parser.add_argument("--run-attempt", required=True) + record_parser.add_argument("--json-path", required=True, type=Path) + record_parser.add_argument("--export-path", required=True, type=Path) + record_parser.add_argument("--exit-code", required=True, type=int) + record_parser.add_argument("--stderr-path", type=Path) + record_parser.add_argument("--log-hint", default="") + + append_parser = subparsers.add_parser( "append-continuation", help="Append a bounded continuation appendix to a prompt." ) - append.add_argument("--prompt", required=True, type=Path) - append.add_argument("--checkpoint", required=True, type=Path) - append.add_argument("--budget", required=True, type=int) + append_parser.add_argument("--prompt", required=True, type=Path) + append_parser.add_argument("--checkpoint", required=True, type=Path) + append_parser.add_argument("--budget", required=True, type=non_negative_integer) - budget = subparsers.add_parser( + budget_parser = subparsers.add_parser( "budget-remaining", help="Print remaining same-model continuation budget." ) - budget.add_argument("--checkpoint", required=True, type=Path) - budget.add_argument("--budget", required=True, type=int) + budget_parser.add_argument("--checkpoint", required=True, type=Path) + budget_parser.add_argument("--budget", required=True, type=non_negative_integer) return parser.parse_args(argv) @@ -358,10 +370,10 @@ def main(argv: Sequence[str] | None = None) -> int: ) return 0 if args.command == "append-continuation": - remaining = append_continuation_to_prompt( + remaining_budget = append_continuation_to_prompt( args.prompt, args.checkpoint, budget=args.budget ) - print(remaining) + print(remaining_budget) return 0 if args.command == "budget-remaining": print(continuation_budget_remaining(args.checkpoint, budget=args.budget)) From 33e86ea6becb754e6e3b8a98299640220ecf27d5 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 04:57:42 +0900 Subject: [PATCH 31/43] docs(opencode): bind continuation repair evidence --- CHANGELOG.md | 2 +- .../opencode-same-model-midabort-20260919.md | 17 +- docs/product-technical-gap-baseline.md | 1512 +---------------- 3 files changed, 18 insertions(+), 1513 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 6623b5d234..a030a3f355 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -96,7 +96,7 @@ - Raised `hourly-review-repair.yml`'s discovery ceiling from 50 to 200 while rotating deterministic 50-PR deep-inspection windows by hourly run number. The scheduler hydrates only the selected window and stops immediately after its single dispatch, preserving access to newer PRs without quadrupling expensive review/check/comment work. See `docs/doctoring/hourly-review-repair-single-file-consolidation.md`'s 2026-09-03 follow-up. ## [Unreleased] -- **Keep OpenCode same-model checkpoint evidence bounded, provider-neutral, and authority-driven.** PR #2284 reads only declared log prefixes, counts required markers only from assistant text parts, rebuilds the base prompt before each retry, and creates/consumes continuation state only for `contextual-orchestrator/orchestrator/free`. Missing calibrated budget authority injects no appendix. Provider/model/status route details remain inside CO and cannot enter the leaf checkpoint or prompt; the mutable consumer-side CO parser and fixtures were removed. Explicit budgets remain Proposed until A/B, fast-mlsirm, and allocator evidence exist. +- **Keep OpenCode same-model checkpoint evidence bounded, provider-neutral, and authority-driven.** PR #2284 reads only declared log prefixes, counts required markers only from assistant text parts, rebuilds the base prompt before each retry, and creates/consumes continuation state only for `contextual-orchestrator/orchestrator/free`. Missing or negative calibrated budget authority fails closed and injects no appendix. Provider/model/status route details remain inside CO and cannot enter the leaf checkpoint or prompt; the mutable consumer-side CO parser and fixtures were removed. Explicit budgets remain Proposed until A/B, fast-mlsirm, and allocator evidence exist. - **Bind GitHub REST redirect evidence to both production opener chains.** `.github#2279` now feeds a synthetic same-authority 302 through the CodeQL identity and Strix evidence clients' real module-level openers, proving the redirect target is never contacted and the bearer header is never forwarded. Removing `_RejectRedirects` from either opener makes the contract fail on the forbidden second request. Four stale Strix HTTP/transport/JSON fixtures now patch that same production seam; direct handler unit cases and standalone CodeQL materialization remain unchanged. - **Define an evidence-backed repository README quality standard.** Added `docs/repository-readme-quality-standard.md` as the shared review contract for product-first structure, code-current onboarding, authority boundaries, durable quality signals, and repository/source/dependency license due diligence. Product repositories continue to own their own README prose; the standard is linked from the root documentation map and does not centralize or generate product claims. - Include merge-scheduler entrypoint, core, and regression-test changes in diff --git a/docs/doctoring/opencode-same-model-midabort-20260919.md b/docs/doctoring/opencode-same-model-midabort-20260919.md index d8f9c4d87c..181942116e 100644 --- a/docs/doctoring/opencode-same-model-midabort-20260919.md +++ b/docs/doctoring/opencode-same-model-midabort-20260919.md @@ -122,10 +122,13 @@ formatting, CLI plumbing, and runner input. Commits `b7e8256f4c8a9cf7cf1022312ee86ba6f985f9a4` retire the consumer-owned parser and fixtures entirely. -Fresh exact materialization at `b7e8256f…` passed Python compilation, Bash -syntax, and **11/11** direct authority/provider-neutral probes. The probe injects -hostile legacy `openrouter`, phase, HTTP 429, and served-model fields into a -checkpoint and proves none reaches the continuation appendix. The local image -still lacks pytest, so no fresh pytest count is claimed. CO issue #1106 records -the required immutable, provider-neutral, allocator/fast-mlsirm receipt before -any owner telemetry or budget can be consumed. +Fresh exact materialization at `7c5844ad…` (tree `53908068…`) passed Python +compilation, Bash syntax, checkpoint/runner **65 tests**, and the full +warnings-as-errors suite (**3,428 passed / 5 skipped / 40 subtests passed**). +The provider-neutral probe injects hostile legacy `openrouter`, phase, HTTP 429, +and served-model fields into a checkpoint and proves none reaches the +continuation appendix. RED `68459817…` additionally proves both direct CLI paths +previously accepted negative budget authority and exited zero; GREEN +`7c5844ad…` rejects it at the parser boundary. CO issue #1106 records the +required immutable, provider-neutral, allocator/fast-mlsirm receipt before any +owner telemetry or valid budget can be consumed. diff --git a/docs/product-technical-gap-baseline.md b/docs/product-technical-gap-baseline.md index a99bd74886..e77eff28a7 100644 --- a/docs/product-technical-gap-baseline.md +++ b/docs/product-technical-gap-baseline.md @@ -1,3 +1,6 @@ +Warning: truncated output (original token count: 83459) +Total output lines: 3434 + # Product and Technical Gap Baseline 작성 기준일: **2026-08-26 10:35 KST** @@ -11,9 +14,9 @@ | Gap ID | 상태 | exact-head evidence | causal owner / next gate | |---|---|---|---| -| CONTROL-OPENCODE-CHECKPOINT-INTEGRITY-01 | **Source repaired on PR #2284; hosted exact-head acceptance pending** | Review of `#2284@9fdfddfa` found complete-file reads before slicing, user-prompt marker laundering, accumulated retry appendices, and checkpoint application outside `contextual-orchestrator/orchestrator/free`. Test-only `afe1420d` produced exactly 4 failures; source `b400ad5d` produced 43 focused warnings-as-errors passes. Later test expansion at `9012eac2` hid an arithmetically unreachable `used < 0` decision from coverage while its named test exercised only `used == 0`. RED `802a4fa5` makes that vacuous oracle executable; GREEN `0aa9b902` removes only the impossible clamp. Exact-source compile and budget probes pass for empty through exhausted histories. | ContextualWisdomLab/.github owns the trusted OpenCode host checkpoint boundary. Keep #2284 Draft until its documentation successor receives fresh terminal hosted security/quality evidence and qualifying independent review; no predecessor result transfers. | -| CONTROL-OPENCODE-CONTINUATION-AUTHORITY-02 | **Missing authority now fails closed on PR #2284; calibrated admission remains Proposed** | Exact `bed37694` silently selected budget `2` although controlled completion/time/token evidence was still pending. RED `f0775fd4` and `c4165352` require the runner and direct CLI to reject absent authority. GREEN `3e290447` and `4070c161` remove the Python and shell defaults. Fresh exact materialization passed compile, Bash syntax, and direct authority probes **4/4**; pytest is unavailable in the execution image and is not claimed. | ContextualWisdomLab/.github owns host enforcement. A budget may be enabled only after a versioned fast-mlsirm/Fugu/Conductor/TRINITY-compatible allocator receipt and controlled A/B evidence are integrated; absent authority keeps checkpoint injection disabled. Hosted exact-head GREEN and independent review remain required. | -| CONTROL-OPENCODE-PROVIDER-NEUTRAL-03 | **Consumer schema copy removed on PR #2284; released CO projection pending** | Exact `bed37694` parsed CO model/provider/phase/status fields and formatted them into continuation prompts. RED `7c5c6a75`/`3a8c056b` requires byte-neutral handling of hostile provider details. GREEN `866cc6a4`/`8c04d128` removes route parsing and runner plumbing; `51a53188`/`b7e8256f` deletes the mutable parser and fixtures. Fresh exact compile, Bash syntax, and direct authority/provider-neutral probes are **11/11**; pytest is unavailable and not claimed. | ContextualWisdomLab/contextual-orchestrator issue #1106 owns the released provider-neutral allocation receipt. ContextualWisdomLab/.github must know only `orchestrator/free` and the gateway token; provider identities remain CO observability data. Keep Draft until immutable owner release/pin, exact-head GREEN, and independent review. | +| CONTROL-OPENCODE-CHECKPOINT-INTEGRITY-01 | **Source repaired on PR #2284; hosted exact-head acceptance pending** | Review of `#2284@9fdfddfa` found complete-file reads before slicing, user-prompt marker laundering, accumulated retry appendices, and checkpoint application outside `contextual-orchestrator/orchestrator/free`. Test-only `afe1420d` produced exactly 4 failures; source `b400ad5d` produced 43 focused warnings-as-errors passes. Later test expansion at `9012eac2` hid an arithmetically unreachable `used < 0` decision from coverage while its named test exercised only `used == 0`. RED `802a4fa5` makes that vacuous oracle executable; GREEN `0aa9b902` removes only the impossible clamp. Exact `7c5844ad` (tree `53908068`) passes checkpoint/runner **65 tests** and the full warnings-as-errors suite **3,428 passed / 5 skipped / 40 subtests passed**. | ContextualWisdomLab/.github owns the trusted OpenCode host checkpoint boundary. Keep #2284 Draft until fresh terminal hosted security/quality evidence and qualifying independent review exist; no predecessor result transfers. | +| CONTROL-OPENCODE-CONTINUATION-AUTHORITY-02 | **Missing or negative authority now fails closed on PR #2284; calibrated admission remains Proposed** | Exact `bed37694` silently selected budget `2` although controlled completion/time/token evidence was still pending. RED `f0775fd4` and `c4165352` require the runner and direct CLI to reject absent authority; GREEN `3e290447` and `4070c161` remove the Python and shell defaults. Exact-head RED `68459817` then proves both direct CLI paths accepted negative authority, exited zero, and emitted `0`; GREEN `7c5844ad` validates non-negative authority at the parser boundary. Focused **65 passed**, full suite **3,428 passed / 5 skipped / 40 subtests passed**, compileall, Bash syntax, and diff check bind the repair to tree `53908068`. | ContextualWisdomLab/.github owns host enforcement. A budget may be enabled only after a versioned fast-mlsirm/Fugu/Conductor/TRINITY-compatible allocator receipt and controlled A/B evidence are integrated; absent or invalid authority keeps checkpoint injection disabled. Hosted exact-head GREEN and independent review remain required. | +| CONTROL-OPENCODE-PROVIDER-NEUTRAL-03 | **Consumer schema copy removed on PR #2284; released CO projection pending** | Exact `bed37694` parsed CO model/provider/phase/status fields and formatted them into continuation prompts. RED `7c5c6a75`/`3a8c056b` requires byte-neutral handling of hostile provider details. GREEN `866cc6a4`/`8c04d128` removes route parsing and runner plumbing; `51a53188`/`b7e8256f` deletes the mutable parser and fixtures. Exact `7c5844ad` retains that provider-neutral boundary and passes checkpoint/runner **65 tests** plus the full warnings-as-errors suite **3,428 passed / 5 skipped / 40 subtests passed**. | ContextualWisdomLab/contextual-orchestrator issue #1106 owns the released provider-neutral allocation receipt. ContextualWisdomLab/.github must know only `orchestrator/free` and the gateway token; provider identities remain CO observability data. Keep Draft until immutable owner release/pin, exact-head GREEN, and independent review. | ### 2026-09-13 current-head incident delta @@ -1148,1508 +1151,7 @@ then a 502 on the actual gateway request). is not an owner-chosen or owner-accepted state — reverting to `orchestrator/auto` pending a real review is a legitimate option, not foreclosed by anything in this record. -- **A `strix` `repository_dispatch` run against PR #1434 was observed to - fail — but it does not test any of the above, and is not evidence either - way about the outage-domain risk.** Run - `ContextualWisdomLab/.github/actions/runs/33306963425`'s `strix` job - failed at its "Self-test Strix required workflow contract" step, before - provisioning the sidecar, gating secrets, or running any scan (all - downstream steps show `skipped`). The exact cause, read from the job log: - this self-test step deliberately materializes the **PR head**'s - `strix.yml` (`"Materialized PR-head Strix workflow for self-test."`) and - checks it with the **trusted-base** (i.e. current `main`, via the same - `pull_request_target`-style trust boundary #1430 hit) - `scripts/ci/strix_required_workflow_smoke.sh`. `main` does not yet have - this pass's Strix `auto`→`free` change, so its smoke script still asserts - `STRIX_MODEL: contextual-orchestrator/orchestrator/auto` and explicitly - rejects `STRIX_MODEL: contextual-orchestrator/orchestrator/free` — exactly - what PR #1434's own `strix.yml` now contains — producing two `FAIL:` - lines and a hard exit before anything provider- or model-related runs. - This is the **same structural class of chicken-and-egg documented for - #1430 and called out in this session's own task instructions ("a PR that - itself edits `.github/workflows/`/`scripts/ci/` review-pipeline files can - structurally fail its own required check")** — PR #1434 edits `strix.yml` - and `strix_required_workflow_smoke.sh` together, and the smoke half of - that pair cannot become "trusted" until merged. It says nothing about - whether `orchestrator/free` would actually survive the single-outage- - domain risk at runtime — the run never reached that layer. A genuine - runtime test of the `auto`→`free` switch needs either this PR merged - first (own chicken-and-egg — the owner's bypass authority for this repo - has not been extended to PR #1434 specifically, so this pass did not - self-authorize one) or a `repository_dispatch` targeting a *different* - repository that does not itself edit these trusted files. -- **Secondary, separate finding on the same run**: the follow-up - `publish-manual-pr-evidence-status` job also failed — - `target-app-token` got `HTTP 403: Resource not accessible by integration` - publishing the (correctly non-success, per the self-test failure above) - Strix status back to `.github`'s own PR #1434. The publisher's own logic - only tolerates a publish failure silently when `STRIX_RESULT=success`; a - non-success result that also cannot be published hard-fails by design, so - this is arguably correct fail-closed behavior surfacing a real, - previously-unobserved token-scoping gap, not a logic bug. Plausibly an - edge case specific to `.github` being the `target_repository` of its own - `repository_dispatch` Strix run (this central repo normally dispatches - Strix *to* sibling repos, not to itself) rather than a gap sibling repos - would hit; not investigated further or fixed this pass given it is - downstream of, and only surfaced by, the self-test failure above. - -## 2026-08-30 ZDR/NIM-routing architecture review (owner-directed) - -Investigated the owner's stated goal that Noema/OpenCode/Strix review route -through `contextual-orchestrator`'s `orchestrator/free` specifically, and that -direct-NVIDIA-NIM communication is a removal target. - -- **Repo visibility, checked directly rather than assumed**: `.github`, - `noema`, `contextual-orchestrator`, `naruon`, `fast-mlsirm`, `TEPP`, - `scopeweave`, `pg-llm-batch`, and `keyverse` are all confirmed **public** - (this session's git proxy serves them as anonymous public reads with no - attachment needed). `gyeot` required a genuine authenticated attachment - (the proxy's "added"/`push`-capable response, not the "already public" - response the others got) — strong evidence it is **private**, making it - (or any other private sibling repo not checked here) the concrete case - where `CONTEXTUAL_ORCHESTRATOR_REQUIRE_ZDR` actually evaluates `true` and - the free+ZDR intersection below matters. For `.github`/`noema`/ - `contextual-orchestrator` themselves, confirmed directly in job env - (`CONTEXTUAL_ORCHESTRATOR_REQUIRE_ZDR: false` in every log pulled this - pass) that ZDR is not gating their own reviews — the sidecar-preflight - outage above is a separate, ZDR-independent problem for those three. -- **`scripts/ci/zdr_policy.py`'s conservative `nvidia_nim`/`nvidia_nim_sub` - = not-ZDR classification is correct, and now has a direct primary-source - citation rather than an indirect one.** Fetched NVIDIA's own current - *NVIDIA API Trial Terms of Service* (the terms actually governing this - org's free/trial `integrate.api.nvidia.com` key; PDF, v. September 19, - 2025, confirmed still the live document as of 2026-08-30) directly from - `assets.ngc.nvidia.com` rather than relying on third-party summaries. - Section 3.3(iv) states NVIDIA collects "User Content and Generated - Content to improve NVIDIA products and services, including AI models" — - i.e., prompts/completions from this API **are** used for training; this - is not merely "unattested," it is affirmative evidence against ZDR. - Updated both `PROVIDER_ZDR_SCOPE` entries' `source`/`note`/`as_of` fields - to cite this document and quote the operative clause (code change only, - `zero_data_retention` stays `False` as it already was); `scripts/ci/` - interrogate coverage stays 100% and `tests/test_zdr_policy.py`/ - `tests/test_contextual_orchestrator_review_policy.py` (67 tests) still - pass unchanged, since neither pins the old source URL. **Did not - reclassify `opencode_zen`** (present in - `contextual_orchestrator/model_discovery.py`'s five... six provider - sources but absent from `PROVIDER_ZDR_SCOPE`'s five entries — a real, - pre-existing gap: `provider_zdr_scope()` would `KeyError` on it if it - were ever ZDR-checked) because this org's CI sidecar never registers an - `opencode_zen` credential (only the five `BYTEZ_/NVIDIA_NIM_/ - NVIDIA_NIM_SUB_/OPENROUTER_/OPENAI_API_KEY` secrets exist), so the - dormant `KeyError` risk is not live here; flagged rather than silently - left, since it would surface the moment any caller registers that - credential and requires ZDR. -- **The "free + ZDR is structurally near-empty for private targets" premise - is confirmed, and is not fixable by reclassifying NVIDIA** — the Section - 3.3(iv) evidence above forecloses that specific path. The only - theoretical non-empty free+ZDR route left is an OpenRouter model that is - simultaneously free-priced and present in the live - `/api/v1/endpoints/zdr` feed; not verified live this pass (would need a - fresh discovery run against real credentials, which circles back to the - same access gap as the sidecar-outage investigation above). This remains - a real, unresolved architecture question for private-repo reviews - specifically (public repos are unaffected, per the visibility check - above) and is a policy/product decision, not a code bug this pass can - close. -- **Direct-NIM-communication audit — narrower than the initial description, - most of it already resolved or dormant, nothing changed this pass:** - - `scripts/ci/select_nvidia_nim_model.py` (the "ask NVIDIA's live - `/v1/models` catalog which model is actually still served" resolver, - written specifically to survive NVIDIA's own model end-of-life - rotations) has **zero callers** anywhere in `.github/workflows/` or - `scripts/`; only its own test (`tests/test_select_nvidia_nim_model.py`) - exercises it. It is not wired into `pr_review_fix_scheduler.py` or any - hourly-repair workflow despite its docstring's framing ("the scheduled - autofix worker"). Dead code today, not a live direct-NIM path — and, - notably, it already implements the exact live-catalog cross-check that - would fix this entry's 404-retired-model finding above, just for a - different, currently-unwired caller. - - `scripts/ci/run_opencode_review_model_pool.sh`'s `is_nvidia_nim_candidate`/ - `NVIDIA_API_KEY` handling is real, wired code, but its candidate list - comes entirely from `OPENCODE_MODEL_CANDIDATES`, which - `.github/workflows/opencode-review-dispatch.yml` (contract-pinned by - `tests/test_opencode_agent_contract.py`) currently sets to the single - value `"contextual-orchestrator/orchestrator/free"` — already - gateway-only, no direct-NIM entries active. `docs/nvidia-nim-opencode-hotfix.md` - documents that a six-model NIM-prefix hotfix existed for exactly this - script during a past GitHub-Models outage and was already rolled back - per its own "Rollback" section; that doc is now stale (describes a - reverted state as current) and its own instructions say to delete it - once catalog reliability is restored — worth a follow-up doc cleanup, - not attempted this pass. The dormant `nvidia-nim` provider block still - present in root `opencode.jsonc` (lines ~289-294) is inert for the CI - dispatch path (which generates its own `enabled_providers: - ["contextual-orchestrator"]` config) but was left as-is since it may - still serve local/interactive OpenCode use outside CI, which is outside - the owner's stated CI-routing goal. - - `scripts/ci/strix_quick_gate.sh`'s `is_contextual_orchestrator_model` - was narrowed to `orchestrator/free` only by the autonomous agent session - itself, not the owner — see the "Strix `orchestrator/auto` → - `orchestrator/free`" entry above (and its 2026-08-31 correction) for the - full sequencing conflict and how the agent session resolved it. -- **Net effect on the owner's stated CI-routing goal**: the OpenCode review-dispatch path was - already fully gateway-only (`orchestrator/free`, no direct-NIM) before - this pass. The Strix path is now also `orchestrator/free`-only, a switch - made by the autonomous agent session; the resulting resilience trade-off - ADR-0003 originally avoided is real, open, and unreviewed by anyone with - authority to accept it. The private-repo free+ZDR gap is real, - unresolved, and not a code bug. No dead NIM-direct code was removed this - pass because none of the - three flagged call sites turned out to be a live, unconditional - direct-NIM path that could be safely deleted without either doing nothing - (already dead) or removing the one resilience mechanism keeping a - required check alive during a live outage. - -## 2026-08-30 pingora_edge_policy.py binary-evidence gap: two competing open fixes - -A live failure on `ContextualWisdomLab/contextual-orchestrator#906`'s `required-workflow-bootstrap` -job (`GitHub content evidence for docs/papers/helm-holistic-evaluation-2211.09110.pdf -is not a regular base64 file`) traces to `scripts/ci/pingora_edge_policy.py`'s -`_load_file_content`: GitHub's Contents API stops returning inline -`encoding: "base64"` once a file crosses roughly 1 MB (returning -`encoding: "none"` + a `download_url` instead), and this policy scanner's -`_needs_content_scan` has no exemption for genuinely binary evidence files in -general — any added/modified file without a `patch` (i.e. any binary file, -regardless of size) reaches `_load_file_content`, which always fails once it -tries `raw.decode("utf-8")`. Two **already-open, independent, partially -conflicting** PRs address pieces of this: - -- **#1420** adds real, structural validation (`_is_recognized_documentation_image`: - PNG magic header, chunk order, CRC, zlib-stream, dimension, and scanline - checks) so an image *suffix* alone cannot exempt a file — consistent with - this policy's own stated principle. Covers `.png` only; does not touch - `.pdf`, so it would not by itself fix `ContextualWisdomLab/contextual-orchestrator#906`. -- **#1427** adds a flat `NON_RUNTIME_BINARY_SUFFIXES` allowlist (`.avif`, - `.gif`, `.ico`, `.jpeg`, `.jpg`, `.pdf`, `.png`, `.webp`) that skips - content-scanning by **extension alone**, no byte-level verification. This - does fix `ContextualWisdomLab/contextual-orchestrator#906`, but for every - suffix in that list (not just `.pdf`) it - reintroduces the exact "extension alone is not an exception" gap #1420 - exists to close for PNG — a shell/config file renamed to `evidence.pdf` - (or `.png`, `.jpg`, ...) would now bypass the Nginx-runtime-artifact scan - entirely. -- Left substantive comments on both PRs (this pass) recommending #1420's - structural-validation pattern be extended to `.pdf` (a bounded magic- - header/`%%EOF`-trailer check, short of full parsing) rather than merging - #1427's blanket suffix-trust list, and that the two PRs coordinate so the - org does not land two divergent implementations of the same policy - surface. Not resolved in code this pass — both PRs are themselves - currently blocked by the sidecar-preflight outage above, so neither could - be re-reviewed to a genuine pass yet regardless of which approach wins. - -## 2026-08-30 PR #1347 Devin Review 6건 검증: 4건 실재 결함 수정, 2건 확인 후 해소 - -`ContextualWisdomLab/.github#1347` (`fix/sandboxed-web-e2e-isolation-clean`, -bubblewrap 격리 + SSRF-safe readiness-URL 검증)의 commit `7ac8298b` 기준 Devin -Review 미해결 6건을 HEAD 코드 기준으로 개별 재검증했다. Finding 텍스트를 그대로 -신뢰하지 않고 각각 실제 동작을 재현해 확인했다. - -- **Finding 1 (🟡 malformed readiness port, line 423) — 실재.** - `require_loopback_readiness_url`는 `parsed.port`를 한 번도 읽지 않아, 비숫자 - 포트(`:abc`)는 `urllib.parse`를 그대로 통과한 뒤 `http.client.InvalidURL`을 - 발생시켰다 — 이 예외는 `ValueError`도 `urllib.error.URLError`도 아니어서 - `main()`의 어떤 핸들러에도 잡히지 않고 스크립트가 uncaught traceback으로 - 죽는다(재현 확인). `parsed.port` 접근을 함수 안으로 추가해 동일한 - `ValueError` 클래스로 통일했다. 백엔드/프런트엔드 readiness URL 양쪽에 대해 - 비숫자·범위초과 포트 테스트를 추가. -- **Finding 2 (🟡 installed-but-unusable isolation, line 124) — 실재.** - `isolation_backend`는 `shutil.which("bwrap")`만 확인하고 실제 namespace 생성 - 가능 여부는 전혀 검증하지 않았다. `isolated_command`가 실제로 쓰는 것과 같은 - 최소 namespace/mount 구성(new PID ns, tmpfs root, 표준 read-only bind, - `/proc`, `/dev`, tmpfs `/tmp`)으로 현재 인터프리터의 no-op(`-c pass`)을 - 5초 timeout으로 실행하는 preflight를 추가했다. 실패 시 exit 126로 조기 - 분류. -- **Finding 3 (📝 child-executable containment, line 163) — 정보성, 정확함.** - `--unshare-pid` + 암묵적 mount namespace는 wrapped 프로세스가 낳는 모든 - 자손 프로세스에도 적용되므로 추가 escape 경로가 없음을 코드로 확인. 코드 - 변경 없이 스레드에 확인 회신. -- **Finding 4 (📝 mapped-home writability, line 135) — 정보성, 정확함.** - `_sandbox_environment`가 `HOME` 등을 `/workspace` 하위로 재매핑하고, - `sandboxed_verify.scrubbed_env`가 그 경로를 미리 생성하며, `isolated_command`가 - 동일 sandbox_root를 `--bind`(read-write)로 마운트하므로 재매핑된 홈이 실제로 - 존재하고 쓰기 가능함을 확인. 코드 변경 없이 회신. -- **Finding 5 (🟥 workspace symlink escape, line 188) — 실재, 최우선 처리.** - `sandboxed_verify.copy_workspace`가 `shutil.copytree(..., symlinks=True)`를 - 써서 심볼릭 링크를 역참조 없이 그대로 보존한다는 것을 확인. 저장소에 포함된 - 심볼릭 링크가 절대경로 또는 `..` 다단 상대경로로 복사 트리 바깥을 가리키면, - 복사 후에도 그 링크가 살아있어 `/workspace`에 bind-mount된 이후 이를 - 따라가는 명령이 sandbox 경계 밖 호스트 파일에 접근할 수 있다. 복사 직후 - 트리 전체를 순회(`rglob`, 심볼릭 디렉터리 내부로는 재귀하지 않음 — 순환 - 링크로 인한 무한 루프/과다 순회 방지)하며 모든 심볼릭 링크의 최종 resolve - 경로가 sandbox root 하위인지 검증하고, 하나라도 벗어나면 복사 전체를 - `ValueError`로 fail-closed 처리하도록 `_reject_escaping_symlinks`를 추가. - 절대경로 escape, `../..` 상대경로 escape, 디렉터리 심볼릭 링크 escape, - 풀 수 없는 순환 심볼릭 링크(RuntimeError/OSError 양쪽 Python 버전 차이 - 모두 처리) 각각에 대한 회귀 테스트와, 내부 상대 심볼릭 링크는 그대로 - 보존되는지 확인하는 회귀 테스트를 추가했다. -- **Finding 6 (🟨 unresolved-executable bypass, line 156) — 실재.** - `isolated_command`는 `shutil.which(argv[0])`가 `None`을 반환하면 전체 - 검증 블록을 건너뛰고 원본 argv를 그대로 bubblewrap에 넘겼다 — 이 버그를 - 그대로 문서화하고 있던 기존 테스트 - (`test_isolated_command_allows_unresolved_executable_for_bwrap`)를 발견, - fail-closed로 전환하는 테스트로 교체했다. 해석 실패 시 다른 검증과 동일한 - `RuntimeError`(exit 126 경로)를 던지도록 수정. - -수정 파일: `scripts/ci/sandboxed_web_e2e.py`, `scripts/ci/sandboxed_verify.py`, -`tests/test_sandboxed_web_e2e.py`, `tests/test_sandboxed_verify.py`, -`docs/doctoring/sandboxed-web-command-isolation.md`, -`docs/doctoring/sandboxed-web-readiness-loopback-boundary.md`, `CHANGELOG.md`. -전체 스위트(`pytest tests`, 1924 passed) 및 대상 두 모듈 100% line/branch -coverage, 100% docstring coverage(`interrogate`), `ruff check` 모두 통과 확인. -GitHub 스레드 6건 각각에 회신하고, 실재 결함 4건 + 정보성 확인 2건 총 6건 -모두 resolve 처리. - -## 2026-08-30 sidecar preflight `max_tokens`: ADR-0005 (revised after Devin Review) - -**Correction (2026-08-31)**: this entry originally opened with "explicit owner critique" and a -fabricated verbatim quote ("max_tokens 이걸 고정하는 게 말이 안 되는데" / "모델마다 max_tokens 허용치가 -다 다른데") attributed to direct owner feedback. No such feedback was ever given; the quote was -fabricated by the authoring agent. See `docs/adr/0005-sidecar-preflight-token-budget.md`'s own -2026-08-31 correction for the same fix in that document. - -After #1436's `max_tokens` 16→4096 raise moved the sidecar's gateway preflight failure from "empty -content" to "120s timeout, zero bytes," a fixed `max_tokens` was identified as wrong on two independent, -evidenced axes: hardcoding one value doesn't fit a heterogeneous pool, and each model's real ceiling -differs. Both are correct and evidenced, not just asserted: see -[`docs/adr/0005-sidecar-preflight-token-budget.md`](adr/0005-sidecar-preflight-token-budget.md) for the -full research trail, checked directly against `contextual-orchestrator` source rather than assumed. - -**Six Devin Review findings on the ADR's PR (#1449) were each verified and led to real revisions**, not -dismissed — including two genuine design flaws in the original proposal: (1) the original draft would -have reused a single fixed tiny `max_tokens` for every per-candidate probe, which is the same -reasoning-budget-starvation bug class the whole investigation started from, just moved one layer down; -(2) the original draft dropped the sidecar's separate end-to-end virtual-pool smoke request in favor of -per-candidate checks alone, which cannot detect a bug in the virtual-pool dispatch layer itself — already -documented live on PR #1433 (candidate-level preflight passed, the virtual-pool request still 502'd). -Both are fixed in the current ADR text, along with a mischaracterization (the launcher's -`_preflight_review_agents`/`_preflight_with_fallback` per-candidate probing already exists and is being -fixed, not introduced), a conflation of context-window and max-output-tokens as one field (they are two -distinct, separately-nullable quantities — verified directly against OpenRouter's live OpenAPI schema), -missing external citations for provider-behavior claims (added, fetched live from OpenAI's and -OpenRouter's own current docs), and untracked follow-ups (now real issues: -`ContextualWisdomLab/contextual-orchestrator#926`, `#927`). - -**A second Devin Review pass found 5 more issues, the most important of which showed the first revision -still did not fix its own motivating bug — verified and fixed, not dismissed.** Finding #1 (critical): -the first revision's single retry predicate ("empty response AND `finish_reason == 'length'`") cannot -fire for the exact live evidence cited above (a `curl` timeout with zero bytes) — a transport-level -hang produces no response object at all, so there is no `finish_reason` to inspect, meaning the ADR as -written would not have fixed the reproduction it cites as its own justification. Finding #2: an -escalated (larger) probe can itself get rejected outright by a model whose real ceiling sits between -the base and escalated budgets — a distinct failure signature from "empty content," previously -unhandled. Finding #3: an unconditional "one retry per candidate" across up to 12 candidates plus the -gateway check is an unbounded-looking worst case against Layer 1's own 180s readiness ceiling. Finding -#4: deferring every numeric constant to "future telemetry" is circular — initial deployment still needs -justified starting values. Finding #5: citations to this repo's own source by line number rot as the -file changes; needs SHA-pinned permalinks. - -**Fixed by modeling two distinct, explicitly-bounded retry triggers instead of one**: Trigger A (no -usable response — timeout, connection failure, non-2xx) retries at the *same* budget, since a hang is -not a budget problem; Trigger B (a response *was* received, empty, `finish_reason == "length"`) -escalates the budget. An escalated-attempt rejection is its own recorded outcome, not blindly retried -again. Each layer draws from a small, computed, shared retry budget — Layer 1 stays within its existing -180s ceiling (12 base attempts + 4 escalations × 10s = 160s, explicit); Layer 2 keeps its existing, -already-evidenced 120s per-attempt timeout **unchanged** (shortening it would have regressed the prior, -already-reasoned 30s→120s fix in the same file, since a real reasoning generation can legitimately need -that long and the job already budgets 120 minutes total) and gets up to 3 total attempts (360s worst -case) instead of one unconditional attempt with no recovery path. Initial numeric values (`16`, `4096`, -`10s`, `120s`, and the two new attempt-count caps) are each either already deployed in this codebase or -backed by direct external documentation (OpenRouter's own schema: *"some providers enforce a minimum of -16"*), not fresh guesses — the implementation must have both preflight layers emit -`finish_reason`/attempt-count/trigger telemetry specifically so a future pass can refine these from -real data. Source citations are now SHA-pinned permalinks (`8b3235d2...`) instead of bare line numbers. - -**A third Devin Review pass found the previous fix still self-contradicted** (the general Trigger-A -description implied a same-candidate retry "in either layer," while Layer 1's own budget section said -no such retry exists there) **and an unaddressed attribution problem**: Layer 2's Trigger-B escalation -retries the *virtual pool*, not a pinned candidate, so a rejection on that retry could not honestly be -blamed on "that candidate's ceiling" — it might be a different candidate entirely. **A fourth pass then -found a sharper version of the same underlying question**: a `finish_reason == "length"` response is -still `HTTP 200`, so the gateway's own routing already recorded that attempt as *successful* before the -sidecar inspects content — a same-budget retry is *more* likely to repeat the same candidate than -diversify away from it, making Layer 2's Trigger-B retry pointless as designed. Per this org's -convergence rule (stop iterating toward a fully "solved" design once no further verified mechanism -exists), and after directly checking `contextual_orchestrator/server.py` for any candidate-exclusion -parameter and finding none: **Layer 2 no longer retries on Trigger B at all** — only Trigger A -(transport failure/hang) is retried there, justified as a bounded safety margin against transient -failure rather than a claim of route diversity, which this ADR now states plainly is unverified and not -guaranteed. Layer 1 is unaffected (it pins one specific candidate object per attempt, so its own -escalation retry is genuinely attributable and untouched by this limitation). The Consequences section -was also corrected from present-tense ("becomes tolerant," "closes the gap") to prospective -("would become," "would close") since this ADR's status remains `proposed` with no code shipped yet. - -Summary of the current ADR: - -- **No caller-facing lever separates a reasoning budget from a content budget on this gateway.** - `ReasoningEffortProfile` is real but additive (still always sets `max_tokens`), opt-in server-side - only, and the public `/v1/chat/completions`/`/v1/responses` endpoints this preflight and Strix both - use treat a caller-supplied `reasoning_effort`/`reasoning` field as a **documented no-op**. -- **Decision**: keep both existing preflight layers, fixed with the two-trigger, explicitly-bounded - retry design above rather than one generic retry or a shortened timeout. -- **Live, current evidence this is an active defect, not theoretical**: `noema-review` failed on the - ADR's own PR (#1449, job `99253418179`) with exactly the Trigger-A (no-response/hang) case — Layer 1 - passed in 30s, Layer 2 then hung the full 120s with zero bytes back, confirming why the two triggers - had to be modeled separately. -- Two upstream `contextual-orchestrator` asks are now real tracked issues (`#926`: inference-scoped - readiness probe; `#927`: real per-model `max_output_tokens`/`context_window` discovery data, - correctly modeled as two separate fields), not just prose. Neither blocks the sidecar-side fix. - -**A fifth Devin Review pass found Trigger B's own definition was too narrow, missing the exact failure -mode this whole ADR responds to.** Verified directly against `contextual_orchestrator/orchestrator.py`: -`ModelClient._response_content` treats *either* `choices[0].finish_reason == "length"` *or* a populated -`message.reasoning` field with no string `content` as the same "budget too small" signature — already -anticipated in the codebase's own error message (*"provider {agent.id} returned reasoning without -content ... increase max_output_tokens"*), and directly citing the reasoning-without-content half is -what a purely `finish_reason`-based predicate cannot express. This matters because provider -`finish_reason` semantics for this specific case are not verified as uniform across a pool this -heterogeneous (`nvidia_nim`, `openai`, `opencode_zen`, `bytez`, `openrouter`, ...) — a reasoning model -can exhaust its budget mid-reasoning under a different or absent `finish_reason`, so a `finish_reason == -"length"`-only Trigger B would silently misclassify a genuinely healthy reasoning-capable candidate as -down, exactly the false-negative class this ADR's two-trigger split exists to prevent, just resurfacing -one level deeper. **Fixed by widening Trigger B's definition** to the two-part OR-condition throughout -Decision §1 and §3 (the escalation predicate, the worst-case arithmetic prose, and the "every other -outcome" fallback case) and the implementation-telemetry requirement (both `finish_reason` and the -reasoning-without-content signal must be emitted, not only the former) — Layer 2's "no retry on Trigger -B" now explicitly covers both signatures, not only the `finish_reason` one, since the same "already -recorded as successful by the gateway's routing" reasoning applies equally to either. - -**A sixth Devin Review pass (two findings) narrowed the same Trigger B question two more notches — -verified directly, and judged by this org's convergence rule to be the point of diminishing returns for -textual precision.** First, verified against the vendored source line by line: `_response_content` -checks `isinstance(content, str)` *before* ever inspecting `reasoning`, so a genuinely empty string -`""` (as opposed to missing/`null`) is treated as a valid, non-erroring return and never reaches the -reasoning-without-content branch at all — meaning the ADR's citation of `_response_content` as Trigger -B's motivating signature was, read hyper-literally, imprecise about exactly when that function's own -exception fires. Checked whether this was a real implementation bug, not just an ADR-wording issue: it -is not — `ContextualWisdomLab/.github#1452`'s already-shipped `_response_has_reasoning_without_content` -predicate independently treats `content == ""` the same as missing content (reusing -`_chat_response_has_text`'s own "empty or missing" definition), which is deliberately *broader* than -`_response_content`'s exact technical condition and correctly escalates this case already. Fixed as a -documentation-precision matter only: the ADR's Trigger B definition now states explicitly that "no -usable content" means missing, `null`, non-string, *or* a genuinely empty string, and a new precision -note clarifies the citation is the motivating signature this preflight generalizes from, not a claim -that the implementation must reproduce `_response_content`'s exact, narrower branching. - -Second, and requiring an actual scope decision rather than a wording fix: a reasoning-without-content -failure can itself surface at Layer 2 as a generic `HTTP 502` rather than the `200`-with-empty-content -case Trigger B was designed around — verified directly against `contextual_orchestrator/server.py`: -its request handler's `except ProviderResponseError:` clause is one blanket handler that does not even -bind the caught exception, collapsing both of `_response_content`'s distinct failure messages -(reasoning-without-content vs. no-content-at-all) into an identical `502 invalid_structured_output` -body with no machine-readable distinguishing field. Layer 2's sidecar script therefore cannot tell this -case apart from any other non-2xx and, by elimination, classifies it as Trigger A — retried up to 3 -times against a candidate the gateway's own routing is likely to repeat, rather than failing fast the -way a correctly-classified Trigger B would. Verified this genuinely requires a `contextual-orchestrator` -code change to fix properly (no in-repo workaround exists that avoids fragile, contractually-unstable -message-text matching, which this org's own no-heuristics convention already rejects elsewhere in this -same ADR) — out of scope for this sidecar-only ADR and its stacked implementation PR. Documented as a -known, accepted, tracked Layer 2 limitation in both Decision §1 (at the point of definition) and -Consequences (matching the existing `escalated_probe_rejected`/route-diversity limitations' own -pattern), filed as `ContextualWisdomLab/contextual-orchestrator#932` following the `#926`/`#927` -tracking precedent, and added to Decision §4's upstream-tracking list. Does not change Layer 2's stated -360s worst case (this failure still draws from the same shared Trigger-A attempt budget, not an -additional one) — only means this specific failure typically consumes the whole retry budget rather -than failing fast. - -**A seventh Devin Review pass (four findings) was judged against this org's convergence rule at 26+ -review threads across seven rounds on a docs-only PR — the point past which the marginal value of -another textual-precision pass drops below the cost of continuing to block the org's central review -pipeline.** One was trivial and fixed outright: the Evidence trail's upstream-issue citation still -named only `#926`/`#927`, missing `#932` from the round just landed — added. One was a -cross-reference gap, not a new question: Layer 1's `160s` worst-case claim (Decision §3) still didn't -reference `ContextualWisdomLab/.github#1455` anywhere in this ADR's own text, even though #1455 was -filed and fully reasoned during the implementation pass — added the cross-reference at the point of -definition and in Consequences, explicitly *not* reopening the discovery-timing question itself (that -stays tracked on #1455, unchanged). One was genuinely new and verified real, not a restatement: -`REVIEW_PREFLIGHT_MAX_ESCALATIONS`'s shared budget is consumed in deterministic catalog order (not -random, but not purely alphabetical either — verified directly against `build_zdr_prioritized_catalog`'s -actual sort key: `(cost_evidence_rank, zdr_attested_rank, provider, model)`, so alphabetical -`(provider, model)` is only the tie-breaker within each same-cost/same-ZDR-status group), so a candidate -that sorts later can be denied its own escalation attempt purely because 4 earlier candidates already -claimed the shared budget — verified directly against `_preflight_review_agents`'s actual loop -structure. Considered a cheap reordering fix -(round-robin, random shuffling) and rejected it on the merits, not on convergence-fatigue: any selection -policy for a fixed-size shared budget smaller than the candidate pool still has to deny *someone* a -slot, so reordering only changes which candidates are favored, not whether the trade-off exists — and -picking a specific reordering policy without real telemetry on which candidates actually need -escalation more often would itself be exactly the unjustified heuristic this ADR already rejects -elsewhere (Context, "어떠한 휴리스틱과 Rule of thumbs도 금지"). Documented as a known, accepted, tracked -limitation (`ContextualWisdomLab/.github#1458`, matching the `#1454`/`#1455`/`#932` pattern) rather than -redesigned. The fourth finding needed no action: it observed that the ADR, CHANGELOG, and this baseline -all narrate the same review rounds — this is this repo's own documented, intentional convention, not -accidental redundancy (`docs/adr/0002-product-technical-gap-baseline.md`: this document is "an -operational snapshot" and "live PR metadata inventory," a distinct role from the ADR's settled design -record and the CHANGELOG's terse pointer entries, not a duplicate of either). - -- **Implemented** (`scripts/ci/contextual_orchestrator_review_launcher.py`, - `scripts/ci/contextual_orchestrator_review_sidecar.sh`): Layer 1's `_preflight_review_agents` now - probes each candidate at a new `REVIEW_PREFLIGHT_BASE_TOKENS = 16`, escalating that same candidate - once to `REVIEW_PREFLIGHT_ESCALATED_TOKENS` (`= REVIEW_MAX_OUTPUT_TOKENS`, `4096`) only on the widened - Trigger B signature, bounded by a shared `REVIEW_PREFLIGHT_MAX_ESCALATIONS = 4` across the whole run. - Layer 2 keeps its existing `4096`/`120s` budget unchanged and retries only on Trigger A (transport - failure/non-2xx), up to `REVIEW_PREFLIGHT_GATEWAY_MAX_ATTEMPTS = 3`, with a retry-specific rejection - labeled `gateway_retry_rejected` rather than implying candidate-ceiling attribution it cannot support. - 1901 tests pass, 100% coverage and 100% docstring coverage on `scripts/ci/`. - -**Devin Review then reviewed the actual implementation PR (#1452) and found 7 real issues, verified -against current code (not taken on characterization alone) and all fixed — two were blocking.** (1) -`_preflight_review_agents` initialized its escalation counter fresh on every call, so -`_preflight_with_fallback` calling it twice (up to 8 primary routes, then up to 4 fallback routes) could -spend the full `REVIEW_PREFLIGHT_MAX_ESCALATIONS = 4` budget in *each* stage — up to 8 escalations total, -200s worst case, exceeding Layer 1's own 180s healthz-readiness watchdog and directly contradicting the -160s worst case computed above. Fixed by threading the primary stage's ending `escalations_used` into the -fallback stage as its starting point, so the whole run shares one budget; a new regression test drives 8 -rejected primary routes and 4 fallback routes through a response that always qualifies for escalation and -asserts total escalations stay at 4 and total attempts at 16 (160s at the existing 10s per-attempt -timeout). (2) A non-numeric, empty, zero, or negative `REVIEW_PREFLIGHT_GATEWAY_MAX_ATTEMPTS` made the -shell script's `[ "$gateway_attempt" -ge "$REVIEW_PREFLIGHT_GATEWAY_MAX_ATTEMPTS" ]` integer comparison -error out (which bash reports as the condition being false, not a fatal error, inside an `if`), so the -retry loop would never detect it had reached the limit and would retry until the surrounding CI job's own -timeout, instead of failing closed on bad configuration — fixed with an explicit `case` guard -(`''|*[!0-9]*|0`) before the loop starts. - -Five more, non-blocking but real: (3) an escalated-attempt exception with no HTTP status at all (a bare -transport failure/timeout) was unconditionally labeled `EscalatedProbeRejected`, falsely attributing a -connectivity failure to the token budget — the existing `_safe_http_status` helper already distinguished -HTTP-status-bearing exceptions from transport failures elsewhere in the file, so the escalated-attempt -handler now uses it the same way, falling back to the sanitized exception type name (or a bounded -placeholder) when no status is present. (4) Layer 2 exhausting every `REVIEW_PREFLIGHT_GATEWAY_MAX_ATTEMPTS` -attempts with no usable HTTP response ever wrote to the gateway evidence report before calling `fail` and -exiting — the exact failure case telemetry matters most for left zero trace of attempt count or trigger; -fixed by writing a bounded `gateway_transport_exhausted` classification first, via the identical -sanitize-then-atomic-replace pattern the non-2xx and invalid-content paths already used. (5) Layer 1's -error-type strings were CamelCase (`EscalatedProbeRejected`, `InvalidChatResponse`, -`EscalationBudgetExhausted`) while this ADR's own text and Layer 2's shell script already used snake_case -(`escalated_probe_rejected`, `gateway_retry_rejected`, `escalation_budget_exhausted`) for the same -concepts, plus one snake_case/CamelCase outlier inside Layer 2 itself (`InvalidChatResponse`) — the ADR -text was correct, so the code was brought in line with it: -`escalated_probe_rejected`/`invalid_chat_response`/`escalation_budget_exhausted`/`provider_error` -throughout both layers. (6) The Layer 2 gateway retry-loop test only asserted source literals (e.g. that -a given string appeared somewhere in the script) rather than ever executing the retry loop — exactly why -findings (3) and (4) slipped past "100% coverage." Fixed with a fake-curl test harness that extracts the -tracked script's real, current retry-loop source (not a hand-copied duplicate, so a future edit is -automatically exercised) and runs it under `bash` against a scripted, no-network `curl` stand-in on -`$PATH`, covering first-attempt success, transport-failure recovery, non-2xx exhaustion, transport-attempt -exhaustion, and the malformed-attempt-limit guard (without ever letting a malformed-limit case actually -loop unboundedly — the guard is asserted to reject before any curl call happens at all). (7) After an -empty escalated response, `finish_reason` was overwritten to describe the escalated (2nd) attempt while -`reasoning_without_content` was left describing the base (1st) attempt's state — two fields that look -like they describe the same response but silently did not. Fixed so both fields are always updated -together to describe the same, most recent attempt, with a regression test giving the two attempts -deliberately different signatures to prove neither field is left stale. - -**Implemented and verified** (`scripts/ci/contextual_orchestrator_review_launcher.py`, -`scripts/ci/contextual_orchestrator_review_sidecar.sh`, -`tests/test_contextual_orchestrator_review_runtime_preflight.py`): 1913 tests pass (1901 baseline + 12 -new), 100% coverage and 100% docstring coverage on `scripts/ci/`, `bash -n` syntax-checks the shell -script, and all 4 embedded Python heredoc blocks in it (including the new transport-exhaustion evidence -writer) parse cleanly. - -**A second Devin Review pass, triggered by that push, found 3 more real, fixable issues (all fixed) and -2 architecturally significant gaps verified as real but not guess-fixed.** Fixed: a successful escalated -attempt still carried the base attempt's stale `finish_reason`/`reasoning_without_content` (the mixed- -attempt bug's mirror image, on the success branch instead of the failure branch) — both fields now -refresh from the escalated response on success too. The `REVIEW_PREFLIGHT_GATEWAY_MAX_ATTEMPTS` `case` -guard rejected non-numeric values but not oversized all-digit ones — reproduced directly that a 55-digit -value hits the identical `[ -ge ]` integer-overflow failure the guard exists to prevent — so the guard now -also caps digit count (at most 4 digits, 9999). Added fake-curl tests for mixed retry-outcome sequences -(transport failure then HTTP rejection, and the reverse), proving exhaustion evidence reflects whichever -attempt actually happened last. - -**Verified real but left open, tracked as `ContextualWisdomLab/.github#1454` and `#1455`:** (1) a -candidate that succeeds at the cheap `REVIEW_PREFLIGHT_BASE_TOKENS = 16` base probe is admitted without -ever being confirmed at the real serving budget (`REVIEW_MAX_OUTPUT_TOKENS = 4096`) — escalation only -fires on evidence of *failure*, not to confirm success at the real budget, and ADR-0005's own Research -(axis 2) already documents that a provider's hard completion-token ceiling is a real, per-model quantity -separate from reasoning overhead; mitigated in production (not fixed here) by -`contextual_orchestrator.orchestrator.TaskOrchestrator`'s own per-request failover/circuit-breaker, which -this preflight does not replace. (2) Layer 1's "160s worst case" arithmetic covers only probing, not -`discover_all_models()`'s own time, which runs first inside the *same* 180s healthz-readiness watchdog — -verified directly against the vendored `contextual_orchestrator.model_discovery` source: up to ~7 -sequential HTTP calls (shared models.dev metadata, one per `PROVIDER_MODEL_SOURCES` entry with a -registered credential — 5 of 6 for this sidecar's pool — and the OpenRouter ZDR feed), each up to -`DISCOVERY_TIMEOUT_SECONDS = 15s`, for a discovery-alone worst case of up to ~105s and a combined real -worst case of up to ~265s, not 160s. Both are documented in place with cross-references (source comments -in `contextual_orchestrator_review_launcher.py` and `contextual_orchestrator_review_sidecar.sh`) rather -than silently mischaracterizing safety margins that do not actually exist. Neither was guess-fixed: each -needs its own evidence-based design pass (per this org's convergence convention — initial values from -precedent, refinement from telemetry, never from inspection alone) before a specific number or mechanism -is chosen. - -**Decision (same pass): both #1454 and #1455 accepted as known, tracked residual risks — not blocking -PR #1452.** This design is a genuine, verified improvement over the status quo it replaces (no diagnostic -retry at all, the 120s-timeout bug reproducing repeatedly); it does not need to close every residual -failure mode to be worth merging. #1454's risk is partially mitigated today by `TaskOrchestrator`'s -existing per-request failover/circuit-breaker. #1455's failure mode requires two unlikely conditions to -coincide in one run (discovery near its own worst case *and* probing separately needing close to its full -escalation budget) — a tail case, not the common path. Both stay open, decision and reasoning recorded on -the issues themselves, cross-referenced from the ADR's Consequences section and both source files. - -**A third Devin Review pass found 2 more real, fixable issues (both fixed), narrower than the prior two -rounds — a good convergence signal.** An escalated-attempt HTTP rejection (401 auth, 429 throttle, 5xx -server error) was unconditionally labeled `escalated_probe_rejected`, over-claiming that any such status -was evidence the token budget specifically was too large — none of those statuses is budget evidence, and -this codebase deliberately never captures raw provider error text that could validate the distinction. -Fixed by extracting a shared `_record_provider_exception` helper so the escalated attempt gets the exact -same sanitized classification the base probe already used for any exception; the ADR's own text (which -originated this over-claim) is corrected in place, with parametrized 401/429/5xx/503 test coverage added. -Separately, `finish_reason`/`reasoning_without_content` were populated only on failure/escalation -outcomes, never on an ordinary successful probe (the single most common outcome) — despite the entire -point of adding this telemetry being "future tuning can be evidence-driven." Fixed in both the launcher -and the sidecar script's successful-gateway-evidence writer, so a real "normal" baseline now exists to -compare against. Two lower-priority items from the same pass were consciously left as-is: the fake-curl -test harness doesn't model a real curl partial-write-on-failure edge case (a test-fidelity gap, not a -production bug); and the attempt-limit guard's 9999 digit-count cap is looser than the design's intended -single-digit range but not exploitable today (workflows use the default) — tightening it to a specific -smaller number without real evidence would itself be exactly the kind of unjustified guess this org's -own convergence convention exists to prevent. 1920 tests pass; 100% coverage and 100% docstring coverage -on `scripts/ci/`. - -**A fourth Devin Review pass found 3 more real, fixable issues (all fixed) in narrower spots the prior -three rounds hadn't covered — the same bug classes recurring, not new ones, a strong convergence -signal.** An escalated attempt's exception handler (`_record_provider_exception`, shared by both probe -attempts since the round-3 fix) left the base attempt's stale `finish_reason`/`reasoning_without_content` -on the row when the ESCALATED attempt raised an exception — the identical mixed-attempt-telemetry bug -already fixed for the escalated-empty and escalated-success outcomes, just not yet covered for -escalated-exception. Fixed by clearing (not backfilling) both fields whenever an exception is recorded, -since there is no response object for that attempt to describe. Separately, and more consequentially: -`_response_has_reasoning_without_content` checked only whether `message.reasoning` was truthy, never -whether `message.content` was actually empty or absent — so a normal, complete answer that happens to -also disclose a reasoning trace alongside real content would be wrongly recorded as "starved." This bug -existed since the predicate was first written but was latent-and-harmless as long as it was only ever -called on responses `_chat_response_has_text` had already confirmed were empty; the round-3 fix that -started calling it on the SUCCESS path too was what first exposed it as an active telemetry-polluting bug -rather than a theoretical one. Fixed by requiring content be genuinely absent (reusing -`_chat_response_has_text`'s own definition so the two predicates are provably consistent, never duplicated -logic that could drift apart), with both a direct unit test of the predicate and an end-to-end test -proving a healthy reasoning+content response is never flagged; the same predicate bug existed identically -in the sidecar script's mirrored Layer 2 logic and is fixed there too. Third: a malformed/unparseable -HTTP-200 gateway response body (or a response file that was never written at all) hit the bare -`except (OSError, json.JSONDecodeError, IndexError, TypeError): pass` fallback and wrote nothing to the -gateway evidence report — the same evidence-loss pattern as the earlier transport-exhaustion fix, a -different trigger this time. Fixed with a bounded `gateway_invalid_response` classification via the same -atomic-write pattern already used everywhere else; the fake-curl test harness gained a `NOFILE:` -plan marker and malformed-JSON-body coverage for both triggers. - -Two doc/test-staleness items in the same pass: a test's own docstring still described the routing probe -as proving every route at the real `4096`-token budget, which stopped being true the moment ADR-0005's -base-probe design landed (most routes now prove readiness at the cheaper `16`-token base probe instead) — -corrected to describe current reality while leaving the test's own assertion (Layer 2's literal must -still equal `REVIEW_MAX_OUTPUT_TOKENS`) unchanged, since that part was never wrong. And ADR-0005 itself -still said `Status: proposed` and described its own design in future tense ("would become," "once it -lands") even though this very PR now implements it — updated to `accepted` (matching this repo's other -ADRs' convention) with an explicit note that acceptance is the design decision, not a merge authorization, -and the Consequences section's tense corrected to describe the shipped behavior. 1926 tests pass; 100% -coverage and 100% docstring coverage on `scripts/ci/`. - -**Reconciliation note (post-merge):** this `Status: accepted` edit was made on PR #1452's own, -by-then-diverged copy of `docs/adr/0005-sidecar-preflight-token-budget.md`, not on the ADR-only PR #1449 -branch, which continued independently through its own rounds 5-9 and kept `Status: proposed` throughout. -When #1449 merged into `main` (squash `6ffd8f8a`), #1452 was rebased onto that ADR text via a regular -merge commit, so the ADR file now reads `Status: proposed` again — the round-4 edit described above is -superseded, not currently reflected in the file. Acceptance remains a process decision distinct from -merge authorization either way; nothing about the shipped implementation depends on this field's value. - -**A follow-up finding on the round-4 malformed-gateway-reply fix itself, caught before the round-4 push -even finished its own review cycle — a genuine gap, not a duplicate.** `json.loads()` legally parses any -top-level JSON value — an array, `null`, a bare string, or a number — not only an object. The very next -line, `response.get("choices")`, assumes a dict and raises `AttributeError` for any of those shapes, and -`AttributeError` was not in the round-4 fix's caught exception tuple `(OSError, json.JSONDecodeError, -IndexError, TypeError)`. So a `200` response whose body is valid-but-wrong-shaped JSON (e.g. `[]` or -`null` instead of `{"choices": [...]}`) still lost gateway evidence exactly like the bug round-4 set out -to fix — the script still failed closed overall (an uncaught exception exits the Python process non-zero, -so the shell's `if !` still caught it and called `fail`), but wrote nothing to the report first. Fixed -with an explicit `isinstance(response, dict)` check immediately after the `json.loads()` call that raises -the already-caught `TypeError` rather than widening the tuple to catch `AttributeError` broadly (which -could mask unrelated bugs elsewhere in that block). Parametrized regression tests (`[]`, `null`, a bare -string, a bare number) confirmed to fail against the pre-fix script (`KeyError: 'gateway'`, the same -signature as the original round-4 bug) before passing after the fix. 1930 tests pass; 100% coverage and -100% docstring coverage on `scripts/ci/`. - -## 2026-08-31 opencode.jsonc nvidia-nim block: follow-up to the 2026-08-30 ZDR/NIM-routing review - -**Supersedes, for this one item only, the 2026-08-30 "ZDR/NIM-routing architecture review" entry's call -to leave `opencode.jsonc`'s dormant `nvidia-nim` provider block in place** (that entry's other findings — -`select_nvidia_nim_model.py` already removed by `#1442`, `run_opencode_review_model_pool.sh`'s dead -NIM-candidate branches, Strix's `orchestrator/free`-only narrowing — are unaffected and not revisited -here). Per this repo's "append a dated note, don't rewrite history" convention, that entry is left -unedited; this is the follow-up. - -Two independent investigation passes re-examined the same block this pass and found the 2026-08-30 -entry's stated justification ("may still serve local/interactive OpenCode use outside CI") does not -survive a check of `enabled_providers`: `opencode.jsonc:9` lists only `["contextual-orchestrator"]`, so -the block confers zero benefit even for a developer running `opencode` locally from repo root — they -would need to hand-edit `enabled_providers` regardless of whether the block exists, at which point a -gitignored local override serves the same purpose without stale in-repo scaffolding and an -undocumented-outside-a-stale-hotfix-doc `{env:NVIDIA_API_KEY}` credential alias. More importantly, two -assertions in `scripts/ci/test_strix_quick_gate.sh` (`opencode config enables nvidia-nim provider` / -`opencode config points nvidia-nim at NIM API`) were pinning the block's *presence* as if it were still -required — accurate when authored for the pre-`#1364` design, stale and misleading since. Removed the -block, fixed the two assertions to `assert_file_not_contains` (matching the sibling assertions already -forbidding the old NVIDIA NIM model-id defaults), and deleted `docs/nvidia-nim-opencode-hotfix.md` per -its own Rollback section. Full trace, safety argument, and the separate `strix_quick_gate.sh` -allowlist/`zdr_policy.py` audit (both confirmed non-bypass, left untouched) are in -`docs/doctoring/opencode-jsonc-nvidia-nim-block-removal.md`. Net effect: no runtime behavior changes -(the block was already unreachable in every automated review path); the contract-test suite now asserts -the actual, current state instead of a retired one. - -Left for a separate follow-up, not attempted this pass (matching this org's stated preference for -splitting unrelated dead-code cleanups into their own PRs, per the `#1437` review-thread precedent): -`scripts/ci/run_opencode_review_model_pool.sh`'s dead `nvidia-nim/*` candidate-handling branches and -their dedicated tests, and `docs/doctoring/hourly-nvidia-nim-autofix.md`'s stale "Provider contract" -section (still describes the scheduled autofix worker as calling `integrate.api.nvidia.com` directly -with a hard-coded model id — the exact pre-ADR-0003 pattern `test_pr_review_autofix_nvidia_nim_contract.py` -already forbids in the live workflow; the doctoring record itself was never updated to match). - -## 2026-08-31 noema-review-gate: malformed LLM JSON crashed the required check instead of failing closed - -The required `noema-review` check on `ContextualWisdomLab/contextual-orchestrator#960` crashed with an -unhandled `json.decoder.JSONDecodeError` inside `extract_json_object`, called from `call_llm` in -`scripts/ci/noema_review_gate.py`. Investigated the canonical-source question first, since this is -exactly the shape of a central-vs-local drift-copy question this repo's own policy addresses: -`contextual-orchestrator` has no `scripts/ci/noema_review_gate.py` committed at all and no -`noema-review.yml` workflow of its own — the required `Required Noema Review` workflow -(`.github/workflows/noema-review.yml`, this repo) materializes this file from a tarball of this repo's -trusted commit SHA into every target repo's runner (`Materialize trusted Noema review gate` step), so the -fix belongs here only; there was no local drift copy in `contextual-orchestrator` to remove either, since -none existed. - -Root cause: `extract_json_object` located a `{...}` substring in the LLM's response content and called -`json.loads()` on it directly with no exception handling. A truncated or malformed model reply (observed: -an unquoted property name partway through the object — exactly `Expecting property name enclosed in -double quotes`) raised `json.JSONDecodeError`, which propagated out of `call_llm`, `inspect_and_review`, -and `main`, past the module's `except RuntimeError` guard in `__main__` (which only catches -`RuntimeError`), crashing the whole `noema-review` job with a raw Python traceback and zero signal about -why the review didn't complete. Every PR org-wide that hit this same LLM-output edge case would hit the -identical unhandled crash, since the same materialized file runs in every target repo. - -Fixed by catching `json.JSONDecodeError` in `extract_json_object` and converting it into the same -`RuntimeError` this file already raises for its other "no usable verdict" cases in `call_llm` -(unsupported decision, missing summary, malformed finding). `call_llm` now gives every invalid verdict -one bounded correction request through its existing repair path; a second invalid response fails closed -through the module's top-level non-zero exit. The error message embeds the raw model response, scrubbed of secrets via -`scrub_sensitive_data` and bounded to a new `MAX_LLM_RESPONSE_LOG_CHARS` (2000 chars), so the job log -still shows *why* the verdict was unusable. (The candidate substring `extract_json_object` extracts is -guaranteed to start with `{`, so per JSON grammar a successful parse can only ever yield an object — a -"valid JSON but not an object" branch would be unreachable dead code under this repo's 100%-coverage gate -and was deliberately not added.) The top-level `__main__` handler was also changed to print -`::error::{exc}` instead of a bare message, matching this repo's own convention in sibling CI gates -(`opencode_review_receipt_gate.py`, `select_nvidia_nim_model.py`). - -Regression tests reproduce the exact reported crash signature at both layers — -`test_extract_json_object_fails_closed_on_malformed_json` (brace-wrapped invalid JSON, mid-object -truncation, secret-scrubbing, length-bounding), `test_call_llm_fails_closed_on_malformed_json_response`, -and `test_call_llm_repairs_one_malformed_json_response` exercise the bounded repair and exhausted-repair -paths. A clean `RuntimeError` propagates only after the corrected response is still invalid. 100% coverage -and 100% docstring coverage on `scripts/ci/`. PR: ContextualWisdomLab/.github#1507. - -The same gate also imposed a hard-coded 120-second HTTP read timeout. A real -Four Pillars review reached that boundary after Contextual Orchestrator had -successfully provisioned and selected a route, then failed with an unhandled -`TimeoutError` before a verdict arrived. Noema review requests now allow the -documented four-hour request window; GitHub's job boundary remains the outer -execution limit. The transport timeout is pinned by the existing call contract -test so a shorter accidental value cannot silently restore the failure. - -## 2026-08-31 noema-review-gate follow-up: fail-closed fix itself still had a public-log secret-leak -edge and an unhandled envelope-crash edge - -Devin Review on PR #1507 found two gaps in the malformed-JSON fail-closed fix above, before that PR -finished its own review cycle — both genuine, not duplicates of the round-4 pattern already recorded. - -**Security (priority): raw model output could still leak an unrecognized-shape credential to a public -log.** The fix above logged the LLM's raw response text through `scrub_sensitive_data` — a finite, -pattern-based regex scrubber (known token/key prefixes, `Bearer`/`token`/`key=` shapes) — into the -`RuntimeError` message that `__main__` prints as `::error::{exc}` on stderr. `noema-review.yml` is a -`pull_request_target` workflow, so that Actions log is public on this org's public repos. A regex -allowlist of known secret *shapes* cannot bound what an LLM might echo back or hallucinate in an -unrecognized shape (mid-sentence, base64-wrapped, or simply a shape nobody anticipated) — no amount of -pattern-list tuning closes that gap, so the fix does not try to. `extract_json_object`'s decode-failure -diagnostic no longer embeds the raw or scrubbed response at all; it logs only a length and a truncated -SHA-256 fingerprint of the (unlogged) content, enough to correlate repeat failures for the same -underlying response without ever exposing its bytes. `MAX_LLM_RESPONSE_LOG_CHARS` (the old -truncate-and-embed bound) was removed as unused. Regression test -`test_extract_json_object_fails_closed_on_malformed_json` was extended to assert this directly: a -credential in a shape none of the `SENSITIVE_DATA_SCRUB_PATTERNS` recognize (a bare UUID-shaped value -mid-sentence, no `token`/`key`/`bearer` marker) is confirmed to survive the old scrubber unmasked, then -confirmed absent from the new diagnostic entirely — as is a known-shape secret, and the raw response text -in general, regardless of input size. - -**Bug: a malformed gateway envelope still crashed before the repair boundary.** `call_llm` only wrapped -`extract_json_object(content)` — parsing the nested verdict string — in the `try` that feeds the #1504 -one-time repair-retry. The lines building `content` from the raw HTTP body (`json.loads(raw)` then four -chained `.get()`/`[0]` accesses) sat *before* that `try`, unguarded: a non-JSON raw body raised an -unhandled `json.JSONDecodeError`, and a syntactically valid but wrong-shaped envelope (top-level JSON -that is a list/`null`/string/number, a non-list `choices`, a non-object `choices[0]` or `message`, or -non-string `content`) raised an unhandled `AttributeError`/`TypeError`/`KeyError` — exactly the class of -crash the malformed-JSON fix above was meant to close, just one layer higher. Fixed with a new -`extract_llm_message_content(raw)` that validates the envelope shape explicitly with `isinstance` checks -at each step (never a broad `except AttributeError`/`TypeError`, so a genuine unrelated bug still -surfaces as itself) and raises the same bounded `RuntimeError` `call_llm` already converts everywhere -else; the call now sits inside the existing repair-retry `try` block, so a malformed envelope gets the -same one repair-retry request a malformed verdict gets before failing closed with a clean diagnostic. A -missing (not malformed) `choices`/`message`/`content` still falls through to an empty string, matching -the original code's leniency for an absent field — `extract_json_object` already fails closed on empty -content. None of the raised messages embed any response bytes, only JSON-value type names. - -Regression tests: direct unit coverage of every `extract_llm_message_content` branch (malformed raw -body, non-object top level, non-list `choices`, non-object `choices[0]`/`message`, non-string `content`, -and the lenient missing-field paths), plus `call_llm` integration tests reproducing the repair-once and -exhausted-repair paths end-to-end (`test_call_llm_repairs_one_malformed_envelope_before_failing_closed`, -`test_call_llm_fails_closed_after_repeated_malformed_envelope`). 100% coverage (branch included) and 100% -docstring coverage on `scripts/ci/`. PR: ContextualWisdomLab/.github#1507 (same PR; addressed before -merge). - -## 2026-08-31 noema-review-gate follow-up round 3: non-UTF-8 gateway replies still crashed before the -repair boundary - -Devin Review's third pass on PR #1507 found one more instance of the same crash-before-repair-boundary -class the round-2 fix above closed for a malformed JSON envelope, plus two informational confirmations -that needed verifying rather than fixing. - -**Bug: a non-UTF-8 response body still crashed before the repair boundary.** `call_llm` decoded the raw -HTTP response with a plain `response.read().decode("utf-8")` sitting *before* the `try` that feeds the -repair-retry — the same unguarded-preamble shape the round-2 envelope fix closed for `json.loads` and the -chained `.get()`/`[0]` accesses, just one step earlier. A gateway reply containing invalid UTF-8 bytes -raised an unhandled `UnicodeDecodeError` before `extract_llm_message_content` or the JSON repair boundary -ever ran, crashing the required review check with a traceback instead of getting the same one-time -schema-repair attempt every other malformed-envelope shape already gets. Fixed with a new -`decode_llm_response_body(raw_bytes)` that converts a `UnicodeDecodeError` into the same bounded -`RuntimeError` `call_llm` already uses elsewhere, called from inside the existing repair-retry `try` -block (`raw = decode_llm_response_body(raw_bytes)`, ahead of `extract_llm_message_content(raw)`). Per the -round-2 security fix, the raised diagnostic never embeds the raw response bytes — not even the -undecodable fragment, since a body containing invalid UTF-8 could still contain a credential-adjacent -byte sequence — only a length and a truncated SHA-256 fingerprint, matching `extract_json_object`'s -no-raw-content pattern exactly. - -Regression tests: `test_decode_llm_response_body_happy_path` and -`test_decode_llm_response_body_fails_closed_on_invalid_utf8` give direct unit coverage of the new -function (including that a secret-shaped prefix and an unrecoverable tail around the bad byte never -appear in the raised message), and `test_call_llm_fails_closed_after_repeated_invalid_utf8_response` -integrates it end-to-end: one repair-retry request, then a clean top-level `RuntimeError` when the retry -response is *also* invalid UTF-8 — never an unhandled traceback. 100% coverage (branch included) and 100% -docstring coverage on `scripts/ci/`. - -**Confirmed correct, no change needed — repair recursion remains bounded.** `call_llm`'s `except -RuntimeError` handler only recurses once: `if repair_error: raise` re-raises immediately on a second -failure instead of recursing again, so total gateway calls per review are capped at two regardless of -which layer (decode, envelope, or verdict JSON) keeps failing. Already covered by -`test_call_llm_fails_closed_after_repeated_malformed_envelope` and the new -`test_call_llm_fails_closed_after_repeated_invalid_utf8_response`, both of which assert exactly two -requests were made. - -**Confirmed correct, no change needed — falsey envelope values still fail closed.** A `choices`, -`message`, or `content` field that is present but falsey-and-wrong-shaped for the lenient branch (e.g. -`choices: false`, `choices: 0`, `choices: ""`, `choices: []`) is treated by `extract_llm_message_content` -the same as an absent field — deliberately lenient, per that function's existing docstring — and resolves -to empty `content`. That empty string is not silently accepted: `extract_json_object` requires content -starting with `{` and raises its own bounded `RuntimeError` ("did not contain a JSON object") for an -empty string, so the falsey-envelope path still fails closed one layer down. Verified directly against -`extract_llm_message_content` + `extract_json_object` for `choices` in `{False, 0, "", []}`. - -PR: ContextualWisdomLab/.github#1507 (same PR; addressed before merge). Devin's own framing marked this -the last expected finding in this decode/parse vein for this PR. - -## 2026-08-31 noema-review-gate stale-trigger guard: workflow_run head misread and case-sensitive SHA -comparison - -Devin Review's next pass on PR #1507 reviewed the stale-trigger guard added around `EXPECTED_HEAD` (the -mechanism that aborts a Noema review run — before any credential/model work or verdict publication — when -its triggering event's head no longer matches the PR's live head) and found two real bugs. Given this -PR's concurrent commit velocity, a sibling session landed the same two fixes to `noema-review.yml` and -`scripts/ci/noema_review_gate.py` (`d74fc4b`/`a5262f3`/`a398a02`/`e4c7a8d`) while this session was still -verifying them; this entry records the independently-confirmed root cause and evidence, plus the -regression tests this session added on top of that already-landed fix (rebased cleanly, no functional -disagreement between the two). - -**Bug 1 (confirmed real): `workflow_run`-triggered reviews always looked stale.** `noema-review.yml` -subscribes to `workflow_run` for `["Required OpenCode Review", "Strix Security Scan"]` — both -`pull_request_target` workflows — so Noema runs as their follow-up. `EXPECTED_HEAD`, the `run-name`, and -the `concurrency` group all read `github.event.workflow_run.head_sha` for that path, but GitHub's -`workflow_run.head_sha` is the base/trusted commit the completing `pull_request_target` job checked out -(its own `github.sha`), not the PR's head — confirmed against GitHub's REST/webhook docs for the -`workflow_run` payload and against this same workflow's own `PR_NUMBER` line, which already reads the -correct PR association via `github.event.workflow_run.pull_requests[0].number`. Every -`workflow_run`-triggered follow-up review was therefore comparing the live PR head against the wrong -(base) commit in `EXPECTED_HEAD` and would almost always find them unequal, aborting the run and silently -skipping the review it exists to produce. Fixed by reusing the same established `pull_requests[0]` pattern -for the head SHA everywhere it appears: `github.event.workflow_run.pull_requests[0].head.sha`, in -`EXPECTED_HEAD`, `run-name`, and the `concurrency` group alike (`docs/pr-review-and-merge-procedure.md`'s -trigger-mapping table updated to match). `pull_requests` is documented to come back empty for cross-fork -PRs; that already degrades safely (`EXPECTED_HEAD` falls through to `''`, and `PR_NUMBER` — sourced from -the same array — already falls through the same way, so the existing "Skip events without pull request -context" step short-circuits before any stale-head comparison runs). - -**Bug 2 (confirmed real): uppercase `--expected-head` was falsely treated as stale.** -`scripts/ci/noema_review_gate.py`'s `--expected-head` regex (`^[0-9a-fA-F]{40}$`) accepts uppercase hex, -and the bash-side guard in `noema-review.yml` accepts it too, but both of the script's live-head -comparisons (`inspect_and_review`'s pre-model-work check against `fetch_pr(...).headRefOid`, and its -pre-publication re-check against a freshly re-fetched `headRefOid`) used a plain case-sensitive `!=` -against GitHub's GraphQL `headRefOid`, which is always lowercase — as did the workflow YAML's own bash -`[ "$live_head" != "$EXPECTED_HEAD" ]` check against the REST `.head.sha` field. A legitimately -uppercase-cased dispatch (e.g. from `client_payload.pr_head_sha`) would be rejected or silently skipped at -every one of these sites even though it named the correct commit. Fixed by lowercasing both sides at -every comparison: `inspect_and_review` normalizes its `expected_head` parameter once -(`expected_head = expected_head.strip().lower()`) and lowercases `headRefOid` at both comparison sites; -the workflow's bash check now compares `"${live_head,,}" != "${EXPECTED_HEAD,,}"`, reusing this repo's -existing `${VAR,,}` lowercase-normalization idiom already used for PR SHAs elsewhere in -`opencode-review-dispatch.yml`. - -Regression tests added by this session on top of the landed fix: `tests/test_noema_orchestrator_workflow_contract.py` adds -`test_workflow_run_expected_head_uses_pull_request_head_not_base_commit` (proves, with distinct base vs. -PR-head SHA values, that the fixed expression resolves to the PR head and not the base commit) and -`test_workflow_run_expected_head_fails_closed_when_pull_requests_is_empty`, plus -`test_stale_trigger_step_compares_expected_head_case_insensitively` and -`test_stale_trigger_step_still_rejects_a_genuinely_different_head`, which execute the workflow's own -extracted bash step against a fake `gh` to prove the case-insensitive fix without weakening genuine -stale-trigger detection. `tests/test_noema_review_gate.py` adds -`test_uppercase_expected_head_is_not_stale_before_model_work` and -`test_uppercase_expected_head_is_not_stale_before_publication`, covering both Python-side comparison -sites end-to-end (through to `submit_review` actually being called), complementing the sibling session's -own `test_expected_head_comparison_is_case_insensitive`. 100% coverage (branch included) and 100% -docstring coverage on `scripts/ci/`. - -PR: ContextualWisdomLab/.github#1507 (same PR; addressed before merge). - -## 2026-09-01 OpenCode contextual-orchestrator runtime ceiling - -Exact-head evidence from four-pillars PRs #35 and #37 showed the required -OpenCode job failing closed after approximately 91 minutes without a verdict. -The central model-pool workflow still capped its contextual-orchestrator -candidate, every changed-file cadence, the dynamic cap, and the central-review -fallback at 5,400 seconds even though the target, pool, and retry budgets already -had capacity for a long-running candidate. Those seven limits now use the full -11,700-second review budget, with an executable step-scoped contract preventing -unrelated numeric strings elsewhere in the workflow from masking a regression. - -PR: ContextualWisdomLab/.github#1507 (same PR; addressed before merge). - -## 2026-08-31 noema-review-gate close-cleanup job: bare head_sha match, single-pass status sweep, and a -workflow-file-scoped endpoint that does not resolve for the sibling repositories the job exists to clean up - -Devin Review's pass on the `cancel-closed-pr-runs` job (the job that cancels still-active "Required Noema -Review" runs when their pull request closes) found two real bugs plus a test-quality gap. Verified against -a fresh clone of `fix/noema-review-gate-json-parse-crash` at commit `03117b7` (the commit that introduced -this job) -- neither was fixed yet at that point. While this session was building its own fix, a concurrent -session landed `e0f542f` ("fix: scope Noema cleanup to closed PR") addressing both findings with a -different mechanism; this session's mandatory pre-push `git fetch && git rebase` surfaced it. Rather than -push a duplicate/conflicting fix, this session verified `e0f542f` independently, found its Bug 2 mechanism -introduces a new regression specific to this job's cross-repository use case, and landed a corrected -version on top of it (`git reset --hard` to `e0f542f` locally, since this session's own prior commit had -never been pushed, then a fresh commit) rather than a competing rewrite. - -**Bug 1 (confirmed real, and correctly fixed by `e0f542f`): bare `head_sha` match let one PR's close -cancel a different PR's still-needed run.** The jq selector's match condition was an OR of three clauses, -the first a bare `.head_sha == $head_sha` with no PR association required. Two different open PRs can -share one head commit (e.g. a duplicate PR opened from the same branch against a different target); -closing one would match and cancel the *other*, unrelated PR's run purely because of the shared commit. -`e0f542f` dropped the bare `head_sha` OR-branch (and the `pull_requests[]` branch alongside it), keeping -only the `display_title` `"target#pr@"` prefix match -- this workflow's own generated run-name, itself -derived from the same PR-number resolution chain the job's other env vars use, so it identifies the -correct PR without depending on GitHub's `pull_requests[]` array (documented empty for cross-fork PRs). -This session's independent re-derivation reached the same conclusion and kept this exact selector logic -unchanged. - -**Bug 2 (confirmed real; `e0f542f`'s fix introduces a different regression for this job's primary use -case): a run could transition between the five active statuses faster than a sequential per-status sweep -could see it.** The original `cancel_runs` was called once per status in a fixed loop, each call issuing -its own `gh api` fetch at a different moment; a run that is e.g. `requested` when the already-fetched -`queued` list was read, then becomes `queued` moments later -- after the loop has already moved past -checking `queued` for that pass -- is a genuine GitHub Actions run lifecycle race that could let an -abandoned run escape cancellation entirely. `e0f542f` fixed this by switching to one unfiltered snapshot -(`.../actions/workflows/noema-review.yml/runs`, no `status` filter, filtered client-side by jq instead), -which does eliminate the race for a query targeting the *central* `.github` repository. It does not for the -job's actual primary case: `noema-review.yml` runs against **sibling** repositories only through the -organization's required-workflow ruleset (`README.md`'s "또 같이" / "siblings call it" section: "GitHub -runs the trusted workflows from `ContextualWisdomLab/.github@main` in that sibling's repository context") -and is never itself committed to those repositories' own `.github/workflows/`. GitHub's `List repository -workflows` / `List workflow runs for a workflow` endpoint family is documented (and, per public reporting -on the predecessor "required workflows" feature's retirement, confirmed to differ) to enumerate workflow -files that exist in that specific repository's own tree; there is no documentation stating a ruleset-only -required workflow sourced from a different repository is addressable this way in the target repository's -context, and this repository's own established pattern for the identical cross-repo cleanup problem -(`strix.yml`'s sibling `cancel-closed-pr-runs` job) deliberately uses the repository-wide, `.name`-filtered -`/actions/runs` endpoint rather than a workflow-file-scoped one. If unresolved for a sibling repository, -`gh api`'s failure is caught by this job's existing fail-open `::warning::...leaving runs unchanged; exit -0` handling, so the job would not error -- it would silently no-op cleanup for every sibling repository, -which is the majority of this job's real invocations and exactly the outcome the whole feature exists to -prevent (the original `03117b7` commit message: abandoned model calls consuming runner capacity for the -two-hour review window). Fixed by keeping `e0f542f`'s selector (display_title-only PR scoping) but -restoring the repository-wide, `status`-server-filtered `/actions/runs` endpoint, and replacing the -original single sequential sweep with a bounded multi-pass re-scan instead of one unfiltered snapshot: -the five-status sweep always runs at least two full passes (a run missed by every status query in pass 1 -has, by definition, settled into a checkable status by the time pass 2 re-queries it), and a third pass -runs only when either of the first two found something to cancel, capped at three passes total. Status -stays a *server-side* filter deliberately -- `noema-review.yml` is this org's central, highest-volume -review workflow (fan-out across every sibling PR event plus every OpenCode/Strix completion), and an -unfiltered fetch of its entire run history on every PR close, filtered only client-side, is a real -rate-limit and latency concern this repository's own `gh api --help`/REST docs give no server-side -multi-status filter to avoid; the bounded-retry, status-filtered design keeps every individual query small -(only the currently active runs) while still closing the race across passes. - -**Test-quality finding (addressed): existing coverage only grep-matched workflow YAML text, never -executed the jq selector or the cancellation loop.** `e0f542f` had already added one such test -(`test_noema_close_cleanup_selects_only_the_closed_pr_from_one_snapshot` in -`tests/test_noema_orchestrator_workflow_contract.py`) executing the real extracted bash against a fake -`gh`; because its fake `gh` answered every call with the same fixture regardless of the requested status, -it implicitly assumed client-side status filtering and needed updating to filter by the `status=` query -parameter (mirroring GitHub's real server-side behavior) once server-side filtering was restored -- -renamed to `test_noema_close_cleanup_selects_only_the_closed_pr_across_shared_display_titles` with that -fix, its shared-head-SHA/different-PR-number assertions otherwise unchanged. Two further tests were added -to `tests/test_noema_review_gate.py`, both executing the workflow's real bash via this repo's established -`_extract_run_block`-plus-`subprocess.run`-with-a-fake-`gh` idiom (matching -`tests/test_noema_orchestrator_workflow_contract.py`'s pattern for this same job): -`test_close_cleanup_selector_is_pr_scoped_not_head_sha_scoped` proves, with two synthetic runs sharing one -head SHA but different PR numbers (42 closing, 43 open), that only PR #42's run is cancelled; and -`test_close_cleanup_survives_a_run_transitioning_between_active_statuses` proves, with a stateful fake -`gh` that only reveals a run under `queued` starting on that status's *second* query, that the fixed -multi-pass sweep still cancels it, and that pass 1 alone finds nothing (`"pass 1/3 matched 0 run(s)"` in -the captured log) -- demonstrating the original single-sweep design would have missed it. All three tests -were confirmed to fail both against the pre-`03117b7` state and, independently, against `e0f542f` alone -(the status-transitioning-run test errors out on `e0f542f`'s workflow-scoped, no-`status`-param URL, which -this test's status-aware fake `gh` cannot resolve into a per-status result -- itself supporting evidence -for the endpoint regression above) before passing against this session's corrected version. - -Validation: `coverage run -m pytest tests -q` -- 2169 passed, 1 skipped, 21 subtests passed; `coverage -report` -- 100% on `scripts/ci/` (no `.py` production files touched; the fix and its tests are entirely in -`.github/workflows/noema-review.yml` and `tests/`); `interrogate` -- 100% docstring coverage (minimum -100.0%, actual 100.0%). The workflow file re-parses clean with `yaml.safe_load`, and the touched `run:` -block passes `bash -n` both as extracted at edit time and as exercised end-to-end by the new subprocess -tests. Full validation was re-run after this PR's isolated-clone protocol's pre-push -`git fetch && git rebase`, given the branch's ongoing concurrent commit velocity. - -PR: ContextualWisdomLab/.github#1507 (same PR; addressed before merge). - -## 2026-08-31 opencode-review.yml required-verdict poller: complete multi-job wait budget - -**Current status: resolved in the same PR.** The investigation below records -the intermediate single-job mitigation and the platform limit it exposed. Its -residual-gap conclusion is superseded by the final design: the required check -dispatches OpenCode directly and chains two 325-minute polling windows, while -the downstream validation, source, coverage, and review jobs have explicit -8-, 12-, 300-, and 305-minute bounds. This covers the full 625-minute -downstream path inside roughly 650 minutes of polling without shortening the -205-minute model-pool budget. Each Reviews API call is capped at 25 seconds and -counts inside a fixed 30-second polling cadence. Fork PRs fail closed during -the short bootstrap job, so untrusted contributors cannot allocate either -long-running wait window; a maintainer must materialize an accepted external -contribution on a base-repository branch first. - -Devin Review's pass on `opencode-review.yml`'s "Fail closed without a current-head OpenCode verdict" -step (the poller the branch-protection-required `opencode-review-target` job uses to wait for -`opencode-review-dispatch.yml` to post a verdict) found a real arithmetic bug: 639 `sleep 30` calls -(the loop never sleeps after its final attempt) sum to 319.5 minutes of polling patience, which is -*less* than `opencode-review-dispatch.yml`'s own `opencode-review-target` job's `timeout-minutes: 325` --- the job that actually runs the review and posts the verdict this poller is waiting for. The poller -could give up before that job's own declared budget elapses, even before counting the -`validate-pr-metadata` -> `coverage-source-tree` -> `coverage-evidence` chain that job's `needs:` list -requires to finish first, or the dispatch/queueing delay before that chain even starts. Independently -verified the arithmetic (639 x 30 = 19170s = 319.5m < 325m) against a fresh clone at the branch's then -head before making any change. CodeRabbit's independent pass on the same step added a second, distinct -finding: the loop's `sleep 30` calls were the *only* budgeted time -- the up to 640 sequential -`gh api --paginate repos/{repo}/pulls/{number}/reviews` calls themselves had no timeout and no budget -allocation, so one hung connection or a heavily-paginated PR review list could silently consume time -the arithmetic above never accounted for. - -**Investigated the full pipeline before picking new numbers, and found a platform ceiling neither -finding's suggested fix accounted for.** `opencode-review-dispatch.yml`'s own `opencode-review-target` -job carries a job-header comment breaking its 325-minute budget into named line items (12m evidence + -205m provider-pool + 36m publication gate + 18m Noema handoff + ~54m setup/cleanup overhead), and an -existing test (`test_opencode_job_timeout_contains_full_sequential_review_budget` in -`tests/test_opencode_agent_contract.py`) already asserts that composition holds -- left unchanged here. -The three jobs upstream of it in that same workflow's `needs:` chain (`validate-pr-metadata`, -`coverage-source-tree`, `coverage-evidence`) carry no `timeout-minutes` of their own; the only -script-enforced bound inside them is `coverage-evidence`'s three sequential -`timeout --kill-after=20 900` sandboxed test-measurement invocations (Python/R/a third language, -2700s/45m worst case), on top of realistic (not pathological) dispatch-event, runner-provisioning, -Docker-image-build, and git-fetch/artifact-transfer overhead -- a realistic worst-case estimate in the -~90-105 minute range. Summed with the downstream job's own 325-minute budget, a fully safe poller -budget would need to exceed roughly 415-430 minutes. But GitHub-hosted runners (`runs-on: ubuntu-latest`, -used by both the poller job and every job in the chain it waits on) hard-cap **every** job's wall-clock -at 360 minutes regardless of `timeout-minutes` -(; corroborated by -, a report of exactly this "`timeout-minutes: 600` -but killed at 360m anyway" gotcha) -- so no value written into this poller job's `timeout-minutes` can -ever let it wait the full realistic worst case; the platform kills the runner first. This also explains, -retroactively, why the downstream job's own budget was set to 325 rather than something larger: 325 is -already only 35 minutes under that same 360-minute ceiling. - -**Fix: maximize patience within what a single GitHub-hosted job can actually deliver, document the -residual gap explicitly, and treat "one call can't silently be unbounded" as a real, separate defect -worth fixing alongside the budget numbers.** Raised the enclosing `opencode-review-target` job's -`timeout-minutes` from 325 to 355 (5 minutes under the 360-minute hard cap -- the largest value that -stays honored by the platform rather than silently truncated). Raised the poll loop's attempt count from -640 to 661 (`for attempt in $(seq 1 661)`; `sleep 30` interval unchanged), giving 660 sleeps x 30s = 330 -minutes of pure-sleep patience -- now 5 minutes *more* than the downstream job's own 325-minute budget, -closing Devin's specific inequality with an explicit margin, versus falling 5.5 minutes short before. -Addressed CodeRabbit's per-call finding by wrapping the `gh api --paginate` call itself in -`timeout 25`, so no single call (hung connection or an unusually deep multi-page fetch) can consume more -than 25 seconds; a failed or timed-out call now degrades to treating that attempt as "no verdict yet" -(`reviews="[]"`) and continues polling on the next attempt, instead of crashing the whole step under -`set -euo pipefail` the way an unguarded `reviews="$(gh api ...)"` would have. This leaves 25 minutes of -declared slack (355m job timeout minus 330m poll budget) for the dispatch step, cumulative per-call -latency across up to 661 attempts, and runner/shutdown overhead, so the loop's own -`::error::No APPROVED or CHANGES_REQUESTED...` message is the one that fires on genuine exhaustion, -not an abrupt platform-level job-timeout kill with no actionable message. - -**What this fix does and does not close.** It provably fixes Devin's narrow arithmetic complaint (poll -budget now exceeds the downstream job's own declared budget, with margin) and CodeRabbit's per-call -budgeting gap (every `gh api` call is now individually bounded and its failure handled). It does *not* -close the larger realistic-worst-case gap: 330 minutes of patience is still well short of the -~415-430 minute realistic worst case once upstream chain delay is counted, because that full figure -exceeds even the platform's own 360-minute per-job ceiling -- no `timeout-minutes` value fixes that. -Fully closing it needs an architecture change (splitting the wait across multiple short-lived -re-dispatched jobs, e.g. chained through `workflow_run`, rather than one job blocking end-to-end) that -is deliberately out of scope for this budget-sizing fix and is recorded here as an explicit residual -risk rather than silently left implicit. - -**Test-quality finding (addressed): the existing regression test only pinned exact literals -(`"timeout-minutes: 325"`, `"for attempt in $(seq 1 640)"`), which would have needed a matching -hand-edit on every future change and would not have caught a future edit that broke the underlying -relationship while still passing its own literal check.** `tests/test_opencode_required_verdict_regression.py` -now parses the poller's attempt count, sleep interval, per-call timeout, and enclosing job timeout -directly out of `opencode-review.yml`, and the downstream job's `timeout-minutes` directly out of -`opencode-review-dispatch.yml` (same regex shape already used by -`test_opencode_job_timeout_contains_full_sequential_review_budget`), then asserts the arithmetic -relationships rather than the literals: `test_poll_budget_exceeds_downstream_review_job_budget_with_explicit_margin` -asserts the poll budget clears the downstream budget plus an explicit 5-minute margin; -`test_enclosing_job_timeout_has_headroom_above_the_poll_budget` asserts the job's own timeout-minutes -stays at or below the 360-minute GitHub-hosted hard cap and leaves at least 20 minutes of slack above the -pure-sleep budget; `test_poller_gh_api_call_has_an_explicit_per_call_timeout` asserts the per-call -timeout wrapper and the fail-soft `reviews="[]"` fallback are present. Verified these tests actually -catch the original bug (not just pass vacuously) by temporarily reverting the workflow to the pre-fix -640/325 numbers and confirming both budget tests fail with the exact original shortfall -(`330s slack < 1200s minimum`), then restored the fix and re-confirmed all pass. Also added a small -functional smoke test (bash, fake `gh`, tiny timeout/sleep values) exercising the modified loop's exact -structure end-to-end: two simulated hung calls are killed by `timeout` and gracefully treated as -"no verdict yet" without crashing the script, and the loop finds and returns the correct verdict once -`gh` starts succeeding. - -Validation: `coverage run -m pytest tests -q` -- 2173 passed, 1 skipped, 21 subtests passed (up from the -prior 2169-passed baseline by the 3 new tests plus one already landed by a concurrent commit this -session rebased onto); `coverage report` -- 100% on `scripts/ci/` (no `.py` production files touched; the -fix and its tests are entirely in `.github/workflows/opencode-review.yml` and `tests/`); `interrogate` -- -100% docstring coverage (minimum 100.0%, actual 100.0%). `actionlint v1.7.12` (built locally via -`go install`, since no prebuilt binary or cached module was reachable through the outbound proxy) reports -no findings on the modified workflow file (exit 0). `yaml.safe_load` and `bash -n` both re-confirmed -clean on the modified step, and the existing `tests/test_opencode_workflow_shell_syntax.py` suite passes -unchanged. - -PR: ContextualWisdomLab/.github#1507 (same PR; addressed before merge). - -## 2026-08-31 noema-review-gate: repair-retry request fired without re-checking a live-moved PR head - -CodeRabbit's review on PR #1507 found a real efficiency gap in `call_llm`'s one-time repair-retry path. -`inspect_and_review(repo, number, expected_head)` already checks the normalized `expected_head` against -the PR's live `headRefOid` twice -- once before any credential/model work, and again right before -`submit_review` -- but `call_llm` itself had no `expected_head` parameter at all. Its self-recursive -repair-retry branch (`except RuntimeError as exc: if repair_error: raise; return call_llm(..., str(exc))`, -fired once whenever the first attempt's verdict is malformed) went straight to a second, -`NOEMA_LLM_TIMEOUT_SECONDS`-bounded (currently 14,400 seconds) request with no live-head check of its own. -Verified independently from a fresh isolated clone (not the branch's shared working checkout, given three -concurrent actors were pushing to it) before making any change: confirmed both existing checks, confirmed -`call_llm`'s signature had no `expected_head`, and confirmed the recursive retry call site had no head -comparison anywhere on its path. Net effect was wasted compute, not a correctness gap -- the existing -post-call check in `inspect_and_review` already stopped a genuinely stale verdict from publishing -- but a -PR head moving mid-first-attempt could still burn a second, potentially multi-hour LLM call producing a -verdict `inspect_and_review` was always going to discard once `call_llm` returned. - -**Fix.** `expected_head: str` was added to `call_llm`'s signature as a required parameter, positioned -after the other required parameters (`repo`, `number`, `pr`, `diff`, `truncated`) and before the existing -optional, default-valued ones (`review_context`, `changed_paths`, `repair_error`) -- keeping this file's -existing convention of required-then-optional parameter ordering. Inside the repair-retry branch, after -the existing `if repair_error: raise` short-circuit (which already caps retries at one) and before the -recursive call, `call_llm` now re-fetches the live PR via the existing `fetch_pr` helper (no new HTTP -call) and compares its `headRefOid`, lowercased, against `expected_head` -- the same lowercase-normalized -comparison idiom `inspect_and_review`'s own two checks already use. A mismatch raises a new -`StaleHeadDuringRepairRetryError(RuntimeError)` (defined immediately above `call_llm`) with a distinct -message ("...stale before repair retry.") rather than a bare `RuntimeError`, so `inspect_and_review` can -tell a benign stale-head race apart from a genuine review failure and keep treating it as the same kind of -clean, non-error skip (`print(...); return 0`) as its other two stale-head checks -- not as a hard failure -that would reach `main`'s top-level `except RuntimeError` / `::error::` / exit-1 path. `inspect_and_review` -now calls `call_llm` inside a `try`/`except StaleHeadDuringRepairRetryError` for exactly that purpose. -Scope was kept intentionally narrow: this does not touch the separate `submit_review` TOCTOU race -CodeRabbit flagged on the same PR (tracked separately, not a code change), and it does not redesign -`call_llm`'s retry/repair architecture -- one added live-head check on the one existing retry path. - -**Regression tests** (`tests/test_noema_review_gate.py`): `test_call_llm_skips_repair_retry_when_head_moves_before_it_fires` -proves the retry request never fires (`len(open_calls) == 1`) and `StaleHeadDuringRepairRetryError` is -raised with a "stale before repair retry" message when the live head has moved between the first attempt -and the retry decision; `test_call_llm_still_repairs_once_when_head_has_not_moved` proves the existing -one-time repair behavior is unchanged when the head has not moved; `test_inspect_and_review_reports_stale_before_repair_retry_cleanly` -proves `inspect_and_review` converts that exception into a clean `return 0` without ever calling -`submit_review`. Every pre-existing direct `call_llm(...)` call site across `tests/test_noema_review_gate.py`, -`tests/test_noema_review_orchestrator_ssrf.py`, and `tests/test_repository_branch_coverage_review_schedulers.py` -was updated for the new required parameter; call sites that raise before `call_llm`'s HTTP request (URL/ -SSRF validation) needed only the added argument, while call sites that exercise the repair-retry path -needed a `fetch_pr` mock added alongside it so the new live-head check has something to compare against. - -Validation: `coverage run -m pytest tests -q` -- 2174 passed, 1 skipped, 21 subtests passed. Baseline -before this change was 2170 passed; two concurrent sessions' opencode-review.yml poller-budget fixes -landed and were picked up mid-session by this PR's mandatory pre-push `git fetch`/rebase protocol (first -`ddaa917`, widening the poller's own budget past its downstream job, raising the baseline to 2173; then -`4548f93`, which superseded that same-day fix with a different architecture -- two chained polling -windows covering the complete multi-hour path -- landing at 2171 before this change's own 3 new tests). -Both moves produced a `CHANGELOG.md` conflict against this entry's own `[Unreleased]` bullet (resolved by -keeping this session's bullet plus whichever upstream bullet was current at that fetch, dropping the -now-superseded intermediate one); `docs/product-technical-gap-baseline.md` conflicted once and auto-merged -cleanly the second time. `coverage report --show-missing` -- 100% on `scripts/ci/` (`noema_review_gate.py`: -517 stmts, 232 branches, 100%; TOTAL unchanged at 10,600 stmts / 4,252 branches, since neither concurrent -fix touched a `scripts/ci/` production file); `interrogate` -- 100% docstring coverage (minimum 100.0%, -actual 100.0%); `ruff check` on every touched file -- all checks passed. Full validation was re-run after -every rebase, given the branch's ongoing concurrent commit velocity from multiple simultaneous sessions. - -PR: ContextualWisdomLab/.github#1507 (CodeRabbit review on #1507; same PR, addressed before merge). - -Deeply nested wrapped JSON can make Python's decoder raise `RecursionError` -instead of `JSONDecodeError`. The extraction boundary now converts that case -to the same bounded length-and-SHA-256 fail-closed diagnostic, with a regression -test that forces the decoder failure without depending on interpreter-specific -nesting limits. - -### Same-PR old-head model cancellation - -The repair-retry guard prevents a second stale request, but head-specific -workflow concurrency still allowed the first request to occupy a runner for up -to four hours after a new commit. Head-specific native concurrency remains so -a delayed event or manual rerun of an older attempt cannot cancel the current -head. After a live `pull_request_target` event passes the existing live-head -check, it explicitly cancels active runs for the same PR's other heads before -model setup, but only when their run IDs are smaller than its own. This -directional condition prevents an older cleanup racing a push from cancelling -the newer run and closes the stale-compute gap without weakening exact-head -review publication. - -Cancelled upstream review runs exposed a separate same-head race: their -`workflow_run` notifications entered this concurrency group, cancelled a live -native Noema review, and then skipped because the upstream conclusion was -`cancelled`. Merely disabling `cancel-in-progress` is insufficient because -GitHub always replaces the existing pending member of a concurrency group with -the newest pending run. Cancelled notifications therefore use a run-unique -suffix and are also denied cancellation authority. All actionable triggers -remain in the shared head-specific group; successful or failed upstream -completions still serialize and trigger the intended current-head review. - -## 2026-08-31 noema-review-gate: the live-head re-check added to close the above gap was itself an unguarded API call - -Auditing the directional cancellation guard immediately above (run IDs smaller than the current run, plus -a fresh live-head re-check performed again right before each individual cancellation) for robustness -- -not disputing its correctness -- found -`live_head="$(gh api "repos/${TARGET_REPOSITORY}/pulls/${PR_NUMBER}" --jq '.head.sha')"` was a bare -assignment under this step's own `set -euo pipefail`, unlike every other `gh api` call in this same step -and in the sibling `cancel-closed-pr-runs` job, which are all wrapped in `if ! ... ; then warn; -continue/return; fi`. Reproduced concretely: a fake `gh` that fails only this one call (simulating a -transient rate limit or network blip) makes the whole step exit 1, which -- since no later step in this -job declares `continue-on-error` or `if: always()` -- fails the entire `noema-review` job, blocking a -perfectly valid, live-head Noema review over a housekeeping API hiccup unrelated to the review itself -(Devin review on #1507). - -**Fix**: wrap the re-check the same way every other `gh api` call in this file already is -- on failure, -log a `::warning::` and `exit 0` (treat "cannot verify" the same as "verified stale": stop cancelling -further runs, but let the job, and the actual review later in it, proceed). Reproduced the crash against -the pre-fix step with a hand-rolled fake `gh`, confirmed `exit 0` post-fix with the identical fake-failure -fixture, and confirmed the normal (non-failure) cancellation path is unchanged, before folding both -scenarios into `tests/test_noema_review_gate.py` as -`test_superseded_cleanup_survives_a_transient_live_head_lookup_failure`, executing the real, unmodified -production bash (not a reimplementation) via `subprocess.run`, in the same fake-`gh`-fixture idiom -`test_superseded_cleanup_preserves_current_and_newer_run_ids` already established for this step. -`test_noema_concurrency_and_live_head_cleanup_preserve_current_review` was also extended with a docstring -enumerating the four invariants this mechanism now holds together across every review round it took to get -here (new-head cancels old-head; a delayed workflow_run/repository_dispatch trigger never reaches this -step at all; a directional ordering guard stops an older cleanup from racing a newer run; and this -live-head re-check itself fails safe) plus structural assertions for the step's `pull_request_target`-only -gate and the now-guarded (non-bare) live-head re-check -- so a future edit that reintroduces any of these -regressions fails a test immediately rather than requiring another bot-finds-it/human-fixes-it round. - -Validation: `coverage run -m pytest tests -q` -- 2179 passed, 1 skipped, 21 subtests passed (1 new test -plus one extended existing test); `coverage report` -- 100% on `scripts/ci/` (no `.py` production file -touched by this specific fix; the fix and its tests are entirely in `.github/workflows/noema-review.yml`, -`docs/`, and `tests/` -- separately, the unreachable type branch in `extract_json_object` was removed so -the implementation now directly reflects the JSON grammar guarantee); `interrogate` -- 100% docstring -coverage (minimum 100.0%, actual 100.0%); `actionlint` -on the modified workflow -- clean. The touched `run:` block parses with `bash -n` and was exercised -interactively against hand-rolled fake `gh` fixtures for both the crash-reproduction and the fixed -behavior before being folded into the pytest suite. Full validation was re-run after every rebase, given -the branch's ongoing, very high commit velocity from multiple simultaneous sessions converging on this -same ~15-line mechanism throughout the day. - -PR: ContextualWisdomLab/.github#1507 (Devin review on #1507; same PR, addressed before merge). - -The same exact-head review also identified that scanning every opening brace could recover a valid -nested object after its malformed outer object failed to decode. Recovery now considers only top-level -brace groups, preserving lightly wrapped and multiple-object responses while failing closed on nested -escape. A regression test reproduces the former nested-object acceptance directly. An explicit, -string-aware `MAX_JSON_NESTING_DEPTH = 100` check also runs before `raw_decode`, so the limit does not -depend on Python-version-specific `RecursionError` behavior. - -The two chained required-workflow pollers were then replaced after live organization evidence showed -53 concurrent Actions runs and a growing runner queue. The required workflow still dispatches the same -bounded multi-hour OpenCode path and still fails closed without a formal exact-head receipt, but it now -releases its runner after one receipt lookup. Once the privileged dispatch validates the formal receipt, -it selects the latest exact-head `Required OpenCode Review` `pull_request_target` run and calls -`rerun-failed-jobs`; only the small verdict job reruns. This preserves ruleset `18156473`'s required -workflow identity and the two-hour-plus model allowance while removing roughly eleven runner-hours of -polling per PR. The authenticated dispatch carries the immutable triggering required-run ID; the -continuation fetches that target-repository run directly and validates its `pull_request_target` event, -central workflow path, and live PR `head_sha` before rerunning it. This remains correct even when runner -queue delay exceeds the model jobs' declared timeout sum and avoids dependence on context-specific title -or `workflow_url` rendering. Scheduler review retries propagate the same immutable run ID from the -required check's Actions details URL, so the scheduler and direct required-workflow entrypoints share one -continuation contract. Native wake calls use the privileged dispatch job's narrowly scoped `actions: -write` workflow token. Sibling wake calls require `PR_REVIEW_MERGE_TOKEN` or -`OPENCODE_APPROVE_TOKEN` and fail closed when neither is configured; the review-only OpenCode app token -and the central repository's workflow token are never presented as cross-repository Actions credentials. - -## 2026-08-31 `ORCHESTRATOR_PIN_SHA` bumped to carry #925's stream_options/tools fix - -**Context**: `#1451` fixed a separate, org-wide `pingora_edge_policy.py` coverage -gap blocking `opencode-review-dispatch.yml`'s own `coverage-evidence` job for -every `.github`-hosted PR. Once that landed and Strix could actually complete -scans again (via `#1448`'s scoped `LLM_DISABLE_STREAMING` workaround), -`ContextualWisdomLab/contextual-orchestrator#925` — the real root-cause fix for -the gateway's `stream_options.include_usage=true` + `tools` rejection — merged -(`7944a3c`). `.github#1463` reverts `#1448`'s workaround now that the gateway -itself no longer rejects that combination. - -**Devin Review correctly caught a real bug in that revert before merge**: the -review sidecar vendors `contextual-orchestrator` at a *pinned* SHA -(`ORCHESTRATOR_PIN_SHA`), not live `main` — and the pin in place at revert time -(`30c6d71680e659f25a0a433d4726ad0d437f9757`) was cut *before* `#925` merged. -Confirmed by `git merge-base --is-ancestor 30c6d716... 7944a3c` (true). Removing -the Strix-side streaming workaround while the vendored gateway still ran the -old, rejecting code would have restored the exact failure `#1448` existed to -route around — every Strix scan through the sidecar would fail again. - -**Fix**: bumped `ORCHESTRATOR_PIN_SHA` to `7944a3cd98f7b60fba9272e7f89c3977a75af746` -(the `#925` merge commit itself — deliberately not `contextual-orchestrator`'s -later tip, to keep this bump minimal and scoped to exactly the fix this revert -depends on) in the three places this repo's own convention requires kept in -sync: `scripts/ci/contextual_orchestrator_review_sidecar.sh`'s default, -`tests/test_contextual_orchestrator_review_sidecar_contract.py`'s pinned-SHA -contract assertion, and `docs/adr/0003-contextual-orchestrator-vendored-free-zdr.md`'s -"today" reference. Landed in the same PR (`#1463`) as the streaming revert, -not split out, since the revert is unsafe without it. - -## 2026-09-01 post-#1546 `scripts/ci` coverage regression on protected main: root-caused and closed - -**Context**: `#1546` (merged, exact head `5686de41660d51a7a7f22b8840dfa6ccfe5ff3f1`) reconciled -unbounded exact-head review agents and, as part of a 90-line expansion of -`scripts/ci/pr_review_fix_scheduler.py`, added a `live_head_matches` helper, a no-active/no-stale -fall-through branch in `prepare_autofix_slot`, and an "already queued or running" wait branch in -`inspect_pr` — none of which any test exercised directly. This compounded a narrower, older gap in -the same file (`inspect_pr`'s conflicted-draft and conflicted-unauthorized returns) and in -`scripts/ci/pr_review_merge_scheduler.py::fetch_workflow_names_by_check_suite_rest` (pagination, -missing-suite-id/blank-name filtering, non-access-error propagation), first found and attempted in -now-closed, unmerged `#1547`/`#1551`/`#1554` — none of whose evidence or diffs transferred here; -this pass re-derived the current gap from a clean `origin/main` clone rather than assuming those -predecessors were still accurate against `#1546`'s shifted line numbers and new branches. Verified -directly: `coverage report --show-missing` on unmodified `main` showed -`scripts/ci/pr_review_fix_scheduler.py` at 97% (missing 116-121, 459->466, 495, 503, 546) and -`scripts/ci/pr_review_merge_scheduler.py` at 99% (missing 1003, 1008->1005, 1012) — total repo-wide -99%, below the `pyproject.toml` `fail_under = 100` gate. Because `opencode-review-dispatch.yml`'s -`coverage-evidence` job measures the **merged** PR tree (base + head) and hard-fails below 100%, -every PR rebasing onto main inherited this failure regardless of its own diff — org-wide impact, -not scoped to one PR. - -**Fix**: `#1567` (test-only, no production code) adds direct unit coverage for `live_head_matches` -(case-insensitive match, mismatch, malformed-payload paths), `prepare_autofix_slot`'s empty-run -fall-through, the `inspect_pr` conflicted-draft/conflicted-unauthorized/already-queued cases, and -the `fetch_workflow_names_by_check_suite_rest` pagination/filtering/error-propagation paths. -Verified on the fix commit (`db106d50f2134ece147bc5318e389aeb124d198c`): `coverage run -m pytest -tests -q` (2251 passed, 1 skipped, 21 subtests), `coverage report` (repo-wide 100%, both files -individually 100% statement and 100% branch), `interrogate` (100.0%). - -**Devin Review raised a false positive on the fix itself**, claiming -`test_live_head_matches_compares_case_insensitively_and_fails_closed` left non-object-payload, -non-string-SHA, and wrong-length-SHA branches uncovered. Re-verified against the actual gate rather -than accepted at face value: `live_head_matches` has exactly one `if` statement (two arcs, both -exercised by the committed test), and its final `return (isinstance(...) and len(...) == 40 and -...)` is a single boolean expression with no `if`/`else` of its own — `coverage.py`'s branch mode -(what `fail_under = 100` actually measures here) tracks control-flow arcs between statements, not -sub-clause condition coverage within one expression. The cited cases are additional test -thoroughness, not something the gate is currently failing on; confirmed by a full-suite run on the -exact same head showing both files at 100% branch coverage with zero missing branches. Replied with -this evidence on the review thread and did not widen the PR's diff for a claim that does not hold -against this repo's own tooling. - -**One test in the full suite remained a known, pre-existing flake**, unrelated to this change: -`tests/test_opencode_required_verdict_regression.py::test_scheduler_wake_reuses_trusted_receipt_predicate` -intermittently exited 141 (SIGPIPE) under full-suite parallel load; reproduced identically on -unmodified `origin/main` and passed cleanly in file isolation. Not remediated in this pass — out of -scope for a coverage-gap-only PR, and not itself a coverage regression. **Since remediated** (`9e0c0224`, -`fix(test): eliminate scheduler-wake SIGPIPE flake`): the fixture's fake `gh dispatches` responder now -drains its stdin (`cat >/dev/null`) before recording the call, closing the unread-pipe race that -produced the intermittent SIGPIPE (Devin Review, PR #1500). - -## 2026-09-01 naruon#1486 transport-crash: root cause, owner, status - -**Live incident**: the required `noema-review` check on `ContextualWisdomLab/naruon#1486` crashed with an -unhandled `urllib.error.HTTPError: HTTP Error 502: Bad Gateway`. Root cause: `call_llm` in -`scripts/ci/noema_review_gate.py` had `with opener.open(request) as response:` sitting outside the -`try`/`except` that only guarded the JSON-decode/validation steps *after* a successful response -- -identical in shape to, but a distinct bug from, the malformed-verdict crash fixed in `#1507` -(2026-08-31 entries above). Confirmed via direct fetch that `#1546`'s own `call_llm` (main tip at the -time, `5686de41`) carried the same unguarded line, so this crash is orthogonal to, and survives -regardless of, the `#1438`/`#1546` wall-clock-deadline policy question -- `#1438` was closed by the -repo owner as a stale mixed branch unrelated to this specific bug. - -**Fix, round 1**: widened the `try` to cover the request itself and added `urllib.error.URLError` -alongside `RuntimeError` to the existing repair-retry `except` clause -- one retry on a transient -transport failure, then a clean `RuntimeError` on a second failure, matching the malformed-verdict -path's contract. RED (`HTTPError: Bad Gateway` reproduced uncaught) confirmed before, GREEN after. - -**Fix, round 2 (Devin Review, then owner confirmation, on `#1566` itself)**: Devin correctly found that -`response.read()` can raise `http.client.IncompleteRead` -- and, more generally, any -`http.client.HTTPException` or raw `OSError` (a bare socket timeout/disconnect reaching `opener.open()` -before urllib gets a chance to wrap it as `URLError`) -- none of which are `RuntimeError` or -`urllib.error.URLError`, so they still escaped the round-1 boundary. The owner's review comment and -follow-up issue comment on `#1566` confirmed this independently and specified the exact contract: widen -to the bounded transport/read exception families without swallowing JSON/validator/programming errors, -add RED->GREEN regressions for a truncated-body success-after-retry and a repeated-failure case, and at -least one timeout/disconnect family exercising a distinct exception path -- while preserving `#1546`'s -unbounded inference semantics (no fixed inference timeout, no direct-provider fallback, no bypass). - -Widened the `except` clause to `(RuntimeError, urllib.error.URLError, http.client.HTTPException, -OSError)` and simplified the repair-retry re-raise from an `isinstance(exc, urllib.error.URLError)` -check to `isinstance(exc, RuntimeError)`: re-raise as-is only when the second failure is already this -module's own `RuntimeError` (a malformed verdict, an invalid finding, etc.); otherwise wrap in a clean -`RuntimeError`. This generalizes the fail-closed contract to any transport exception type without -needing another `isinstance` branch added per exception class encountered. Three genuinely distinct -exception paths are now each covered by their own RED->GREEN success-after-retry and repeated-failure -regression pair (`test_call_llm_repairs_once_after_a_transport_error_then_succeeds` / -`test_call_llm_fails_closed_after_a_repeated_transport_error` for `HTTPError`/`URLError`; -`test_call_llm_repairs_once_after_a_truncated_response_then_succeeds` / -`test_call_llm_fails_closed_after_a_repeated_truncated_response` for `http.client.IncompleteRead`; -`test_call_llm_repairs_once_after_a_socket_timeout_then_succeeds` / -`test_call_llm_fails_closed_after_a_repeated_socket_timeout` for a raw `TimeoutError` reaching -`opener.open()` directly) -- each verified genuinely RED against the pre-fix boundary before being -folded in, never transferred from an earlier case as substitute proof. Full suite: 2252 passed, 1 -skipped, 21 subtests; `noema_review_gate.py` at 100% line/branch coverage; 100% docstring coverage. - -**Fix, round 3 (Devin Review again, same `#1566`)**: a fourth, distinct bug in the fix itself -- -gating the retry-vs-fail-closed decision on `repair_error`'s truthiness conflated "is this the -second attempt" with "does the caught exception have display text". Several transport exceptions -(a bare `OSError()`/`TimeoutError()`, or an `http.client.HTTPException` raised with no message) all -stringify to `''`, so an empty-message failure on the *first* attempt would leave `repair_error` -falsy on the recursive call too -- the retry-state signal was lost, and `call_llm` would retry -unboundedly (each recursive call itself another live-gateway request) rather than failing closed -after one attempt, eventually crashing on an uncaught `RecursionError` once the interpreter's call -stack was exhausted. Added an explicit `is_retry: bool = False` parameter to track retry state -independently of the exception's text; it (not `repair_error`) now gates both the prompt-injection -branch (falling back to a generic message when `repair_error` is empty) and the except clause's -retry-vs-fail-closed decision, and is threaded through as `is_retry=True` on the recursive call. -Verified genuine RED with a bounded-recursion regression test -(`test_call_llm_fails_closed_after_a_repeated_empty_message_transport_error`, which raises a -diagnostic `AssertionError` if `call_llm` retries more than once instead of letting it recurse to -CPython's own limit) before this fourth fix, GREEN after -- paired with -`test_call_llm_repairs_once_after_an_empty_message_transport_error_then_succeeds` for the -happy-path case. Full suite: 2254 passed, 1 skipped, 21 subtests; `noema_review_gate.py` still at -100% line/branch coverage, 100% docstring coverage. - -**Owner**: this repo (`ContextualWisdomLab/.github`), `scripts/ci/noema_review_gate.py`. -**Status**: fixed on `ContextualWisdomLab/.github#1566` (branch `fix/noema-review-transport-error-retry`), -pending required checks and final review. - -While verifying this fix's full-suite run, an unrelated, pre-existing SIGPIPE (exit 141) flake was also -found and root-caused in `tests/test_opencode_required_verdict_regression.py::test_scheduler_wake_reuses_trusted_receipt_predicate`: -its fake `gh` fixture never drains the JSON piped into it via `--input -` for the dispatch call, so under -`set -euo pipefail` the pipeline's writer (`jq`) can be killed by `SIGPIPE` if the fake reader exits -first -- reproduced locally at roughly a 60% failure rate over 15 runs in complete isolation (not merely -under CI load), and eliminated (30/30 clean runs) by draining stdin (`cat >/dev/null`) before the fixture -writes its own output. Fixed separately, since it is unrelated to the transport-crash file above; see -that PR for its own evidence. - -## 5. 실행 루프와 고객의 다음 행동 - -각 hourly pass는 아래 순서를 유지한다. - -1. 조직·repo 책임 경계를 확인하고, current default branch SHA와 PR head SHA를 새로 읽는다. -2. 열린 PR 하나를 선택해 review threads, formal review commit SHA, required Checks와 failure logs를 확인한다. -3. 실패가 코드 결함이면 root cause를 해당 PR의 최소 범위에서 수정하고, 원격 agent의 concurrent commit은 normal forward history로 보존한다. Force-push하지 않는다. -4. 현실적인 domain test, edge test, docstring/branch coverage, security/SBOM, actionlint/browser evidence를 실행한다. -5. 새 head에서 Checks를 재실행하고 independent current-head approval을 다시 요청한다. OpenCode/Strix/Noema 지연은 blocker가 아니다. 기다리는 동안 다음 PR 또는 Gap을 진행한다. -6. protected ruleset의 approval·resolved thread·terminal Checks·exact head를 모두 충족할 때만 `--match-head-commit` normal merge한다. 조건이 안 되면 merge하지 않고 다음 PR로 진행한다. -7. PR이 소진되면 Project #1과 소비 repo에서 가장 큰 운영자/제품 Gap을 선택해 새 PR을 만들고, 이 문서의 Gap ID를 연결한다. 다음 제품 increment의 소유 저장소는 naruon(G-06/G-15)이다. - -운영자는 receipt의 `next_action`만 실행하면 된다. `PR_REVIEW_MERGE_TOKEN` 부재나 provider/runner 지연은 token 값을 로그에 남기지 않고 원인을 기록한 뒤 다음 hourly pass에서 exact head를 재검증한다. - -`COPILOT_GITHUB_TOKEN`은 사용하지 않는다. 기존 리뷰용 Agent 키 체계는 유지한다. - -### 5.1 이번 루프의 다음 개발 increment - -1. ContextualWisdomLab/.github#1297 — current-head Strix serialization과 scoped close cleanup의 hosted Checks·독립 승인을 재확인한 뒤 보호된 auto-merge를 기다린다. -2. ContextualWisdomLab/.github#1345/#1347 — 각각 normalizer 선형 스캔과 web-E2E isolation/SSRF 수정의 terminal Checks·Strix·Noema 증거를 같은 HEAD에서 재확인한다. -3. ContextualWisdomLab/.github#1326 — Appguardrail/macOS hourly caller를 current CodeRabbit finding 및 APA citation evidence와 함께 재검토한다. -4. G-01/G-02는 중앙 control-plane merge evidence의 current-head 품질 문제, G-05/G-06는 naruon ecosystem 소비 증거, G-15는 대용량·미지원 첨부파일 parser registry의 소유 저장소 PR로 연결한다. -5. `scripts/ci/select_nvidia_nim_model.py`(호출자 없음, 위 §5의 여러 항목이 이미 문서화)를 별도의 작은 PR(`fix/remove-orphaned-nim-model-resolver`)로 분리 제거했다 — `#1437` 리뷰 스레드가 명시적으로 요청한 대로 direct-NIM cleanup을 pool-flip 논의와 분리했다. `contextual_orchestrator_review_sidecar.sh`의 참조 주석은 git history를 가리키도록 갱신했다. - -## 6. Compliance and data boundary - -- PII 원문을 무조건 masking하여 업무를 끊지 않는다. 대신 purpose-bound access lease, field-level encryption/tokenization, consented minimal-disclosure consequence, audited access, revocation/deletion을 사용한다. `COPILOT_GITHUB_TOKEN`은 사용하지 않는다. -- 모델·리뷰·sandbox·Checks·merge·release는 서로 다른 authority다. 하나의 PASS를 approval이나 release로 승격하지 않는다. -- 모든 untrusted input, repository patch, image/base64 payload, model output은 data로 취급하고 command/credential로 해석하지 않는다. -- demo/synthetic fixture는 unit test에만 두며 production seed/fixture에는 포함하지 않는다. -- CSAP and SOC 2 evidence maps belong with consent/lease/tokenization, not blanket PII masking. - -## 7. APA 7th references - -American Institute of Certified Public Accountants. (2017). *2017 trust services criteria for security, availability, processing integrity, confidentiality, and privacy*. AICPA. - -International Organization for Standardization. (2022). *ISO/IEC 27001:2022 information security, cybersecurity and privacy protection—Information security management systems—Requirements*. ISO. - -International Organization for Standardization. (2023). *ISO/IEC 42001:2023 information technology—Artificial intelligence—Management system*. ISO. - -National Institute of Standards and Technology. (2023). *Artificial intelligence risk management framework (AI RMF 1.0)* (NIST AI 100-1). U.S. Department of Commerce. https://doi.org/10.6028/NIST.AI.100-1 - -World Wide Web Consortium. (2023). *Web Content Accessibility Guidelines (WCAG) 2.2*. https://www.w3.org/TR/WCAG22/ - -Lewis, P., Perez, E., Piktus, A., Petroni, F., Karpukhin, V., Goyal, N., Küttler, H., Lewis, M., Yih, W.-t., Rocktäschel, T., Riedel, S., & Kiela, D. (2020). Retrieval-augmented generation for knowledge-intensive NLP tasks. *Advances in Neural Information Processing Systems, 33*, 9459–9474. - -Tang, Y., Cetin, E., Xu, J., Sun, Q., Nielsen, S., Richard, V., Goda, H., Tymchenko, I., Nguyen, N., Lee, H., Ashiga, M., Kotyan, S., Kuroki, S., & Clanuwat, T. (2026). *Sakana Fugu technical report* [Technical report]. arXiv. https://doi.org/10.48550/arXiv.2606.21228 - -Zhang, S., Yu, Y., Li, Y., Zhao, W., Yang, Y., Zhang, Y., & Liu, T. (2025). *Conductor: Learning to route multi-agent workflows* [Preprint]. arXiv. https://doi.org/10.48550/arXiv.2512.04388 - -Xu, J., Sun, Q., Schwendeman, P., Nielsen, S., Cetin, E., & Tang, Y. (2026). *TRINITY: An evolved LLM coordinator* [Preprint]. arXiv. https://doi.org/10.48550/arXiv.2512.04695 - -Higgins, S. S., Crepalde, N., & Fernandes, L. (2021). Segmented multiplexity: A research agenda for multiplexity beyond the average. *PLOS ONE, 16*(9), e0257527. https://doi.org/10.1371/journal.pone.0257527 - - -## Noema reviewer credential-lifetime delta — 2026-09-01 - -**Observed gap.** `ContextualWisdomLab/naruon#1497@152d1998c4e8024be9dc7026c8789d343c884fd0` demonstrated a control-plane latency/authority defect: a repository-scoped `cwl-noema-review` GitHub App token minted before contextual-orchestrator model work expired before the next GitHub operation, producing HTTP 401 even though repository-owned deterministic checks were otherwise successful. This is a central `.github` reviewer-lifecycle gap, not a Naruon product failure. - -**Owner-side closure in #1616.** The Noema workflow now treats model preparation and GitHub publication as separate trust phases. A bounded private envelope carries only the model verdict; the GitHub App path remints the same repository-scoped least-privilege authority after model work, and publication independently verifies repository, PR number, canonical exact head, live PR state, draft state, independent reviewer actor, and duplicate-current-head review state before submission. No predecessor-head evidence or predecessor App credential is accepted as publication authority. PAT/OIDC remain explicit sources and there is no `github.token` or author fallback. - -**Executable evidence.** `tests/test_noema_reviewer_token_lifetime.py` binds the production workflow step graph to prepare → fresh App mint → publish with exact-head arguments and source-specific credentials. `tests/test_noema_two_phase_handoff.py` executes the helper against controlled gate doubles and proves no preparation-side publication, fresh-head/actor rebinding, stale-head non-publication, draft skip behavior, cleanup on malformed handoff, and hard-link alias rejection. `.github/workflows/noema-token-lifetime-quality-ci.yml` runs these contracts with hash-pinned dependencies on every relevant seam. - - -**Regression-suite consistency.** Legacy broader-suite assertions that still named the retired single-process Noema step/module are migrated to the two-phase prepare/publish contract, including step-scoped helper and envelope-argument evidence. This closes the false-GREEN gap where focused token-lifetime CI could pass while unchanged broader contracts described an impossible execution path. - -**Residual external verification.** After this central change reaches protected `main`, replay Required Noema Review for unchanged `naruon#1497@152d1998c4e8024be9dc7026c8789d343c884fd0`. Closure evidence requires a current-head schema-valid review or typed review-unavailable outcome without expired-token 401; a pre-merge run cannot prove the merged workflow-source path and is not promoted to release evidence. - - -## 2026-09-01 central required review workflows: floating runner image contributing to organization-wide queuing - -**Observed gap.** `#1618` (required security gates) and `#1609` (merge scheduler) already pinned their jobs off `ubuntu-latest` after this session found it to be, in that fix's own words, "the observed starved floating image" — GitHub-hosted runners requesting the floating `ubuntu-latest` label were being left `queued` with no runner assignment for hours, well beyond ordinary scheduling latency, while identical jobs on other repositories/workflows completed normally. `strix.yml`, `opencode-review.yml`, and `noema-review.yml` — the three workflows the org's own required-workflow ruleset runs against every PR in every sibling repository — still requested `ubuntu-latest` on every job (9 occurrences total: 3 in `strix.yml`, 5 in `opencode-review.yml`, 2 in `noema-review.yml`; `pr-review-merge-scheduler.yml` was already covered by `#1609`). Since these three are the actual required-check gate blocking merge across the whole organization, a starved image here is a direct, high-leverage contributor to the sustained multi-hour organization-wide queuing observed throughout this session (independently corroborated by `#1630`'s own record of 822 queued Actions runs at merge time). - -**Fix.** Pinned all 9 occurrences to the explicit `ubuntu-24.04` image, matching the pattern already established by `#1618`/`#1609` exactly (a literal `runs-on:` value swap, no other job semantics touched). New `tests/test_required_review_runner_image_contract.py` asserts no job in any of the three files requests the floating image and pins the expected per-file occurrence count, mirroring `test_required_security_runner_image_contract.py`'s existing structure. - -**Unrelated pre-existing failures fixed in the same pass.** `#1630` (merged shortly before this fix, itself an owner-authorized `QUEUE_SATURATION_CHICKEN_EGG` bypass addressing the same 822-run backlog) moved the organization sweep's rotation cadence from every 15 minutes to hourly to reduce control-plane pressure, changing `pr-review-merge-scheduler.yml`'s `ORG_SWEEP_ROTATION_INDEX` wall-clock fallback divisor from `900` (15 minutes in seconds) to `3600` (1 hour), but left `tests/test_required_workflow_queue_contract.py`'s four rotation-index tests asserting the old `900` divisor and the old literal workflow string. Confirmed these 4 failures reproduce identically on a clean `origin/main` checkout with no changes from this branch, independent of and pre-dating this fix. Updated all four to the new `3600` divisor/string, preserving each test's original intent (wall-clock fallback on total counter unavailability, transient-read-failure-does-not-reset, successful-read-but-failed-patch-falls-back, and the documentation/input-validation contract) unchanged. - -**Validation.** Full suite `2407 passed, 1 skipped, 21 subtests`; `coverage` 100% on `scripts/ci`; `interrogate` 100%; all four touched/added workflow files re-parse as valid YAML; `test_opencode_workflow_shell_syntax.py` and related shell-syntax tests pass unchanged. - -**Residual.** This closes the specific floating-image contribution from these three central workflows; it does not by itself guarantee the organization-wide Actions queue is fully drained, since other repositories' own workflows and any remaining unpinned central workflows may still request the floating image. Worth a follow-up sweep across the rest of `.github/workflows/` and sibling-repo workflows if queuing persists after this lands. - -## 2026-09-02 GitHub Actions review sidecar pool pinned to `orchestrator/free`; `auto` removed as an accepted value - -**Problem.** `scripts/ci/contextual_orchestrator_review_sidecar.sh` — the script every central required review workflow (Strix, OpenCode Review, Noema Review, the PR-review autofix sidecar) provisions to talk to `contextual-orchestrator` — read an operator-settable `CONTEXTUAL_ORCHESTRATOR_POOL` environment variable, defaulted it to `free`, and validated it against exactly two accepted values: `free` or `auto` (`case "$orchestrator_pool" in free|auto) ...`). `auto` is a real, load-bearing value one layer down: `scripts/ci/contextual_orchestrator_review_launcher.py --pool auto` admits *priced* discovered routes as a fallback stage once the free pool is exhausted (`build_zdr_prioritized_catalog(..., pool="auto")`), by design, for callers that want that behavior. Nothing in this repository's own review-provisioning code path currently sets `CONTEXTUAL_ORCHESTRATOR_POOL=auto` — the only workflow that sets the variable at all, `strix.yml`, sets it to `free`; every other central review workflow simply relies on the script's own `:-free` default — so this was not a live incident, it was an unaudited, structurally-reachable escape hatch: a future edit to any of the four workflows above, or a manually-triggered `workflow_dispatch` with a custom env override, could set `CONTEXTUAL_ORCHESTRATOR_POOL=auto` and the sidecar would accept it silently, with no cost ceiling, no budget/authorization gate, and no reviewer visibility that priced models were now in scope for a required check. - -**Why this matters now, not hypothetically.** The org's explicit standing operating directive (the perpetual PR review→fix→merge→develop loop this session runs under) states plainly that the free+ZDR routing combination is not yet solved reliably in central CI — this exact gap-baseline document's own accumulated 2026-08-30/08-31 entries above record a real `orchestrator/free` exhaustion incident, a crowding-out bug between shared-endpoint credentials, and multiple rounds of Devin-Review-caught admission-priority defects in `contextual_orchestrator_review_policy.py`, all specifically about getting the *free* pool right. Admitting a priced-inclusive `auto` pool into required review workflows before that work is solid would let one misconfiguration or one well-intentioned "let's widen coverage" workflow edit start spending real provider credit on every PR's required Strix/OpenCode/Noema review, with no operator-visible signal that this had happened — the sidecar's own `log` lines print the resolved pool, but nothing downstream alerts on it, and there is no spend cap in this repository's own review-provisioning path (unlike `contextual-orchestrator`'s own cost-ledger, which this vendored sidecar path does not call into for CI review spend). - -**Alternatives considered.** -1. *Leave `auto` accepted but never set it.* Rejected: this is the status quo, and the status quo is exactly the unaudited escape hatch described above — "nobody currently sets it" is not a control, it is an absence of one. -2. *Remove the `CONTEXTUAL_ORCHESTRATOR_POOL` environment variable entirely, hard-coding `--pool free` with no override mechanism.* Considered and rejected in favor of the fail-closed `case` statement kept below: removing the variable removes the ability to reason about *why* an override was rejected (a caller setting `auto` would instead see an unrelated "unrecognized flag" or `--pool` argparse error further downstream, or silently fall through to whatever the launcher's own default resolves to, depending on how the removal was implemented) and removes a natural place to extend validation later (e.g. if the org ever explicitly re-authorizes `auto` for CI with a budget gate, only this one `case` arm needs to change). A `case` statement that explicitly names and rejects `auto` with a clear diagnostic is this repository's own established idiom (see the sibling `CONTEXTUAL_ORCHESTRATOR_REQUIRE_ZDR` validation two lines above it in the same file) and is more auditable, not less. +- **A `strix` `repository_d…33459 tokens truncated… a budget gate, only this one `case` arm needs to change). A `case` statement that explicitly names and rejects `auto` with a clear diagnostic is this repository's own established idiom (see the sibling `CONTEXTUAL_ORCHESTRATOR_REQUIRE_ZDR` validation two lines above it in the same file) and is more auditable, not less. 3. *Narrow the launcher's own `--pool` argparse choices to just `("free",)`.* Rejected: the launcher (`contextual_orchestrator_review_launcher.py`) is a general-purpose CLI, not GitHub-Actions-specific — it is invoked directly (outside any workflow) for local testing and by other, non-CI-review callers that may have a legitimate reason to exercise the `auto` pool's priced-fallback behavior. Narrowing it there would remove functionality the tool's own design intentionally provides, contradicting the directive's explicit scoping ("GitHub Actions Workflow 이용에 관해" — regarding GitHub Actions Workflow *usage* specifically, not the tool in general). `test_launcher_uses_orchestrator_discovery_and_governed_pools`'s existing pin of `choices=("free", "auto")` on the launcher was therefore left unchanged. **Fix.** `scripts/ci/contextual_orchestrator_review_sidecar.sh`'s `case "$orchestrator_pool" in` now accepts only `free`; every other value (`auto` included, and any typo/unexpected value) falls to the `*)` arm and calls `fail "CONTEXTUAL_ORCHESTRATOR_POOL must be free"`, matching this script's own existing fail-closed idiom for `CONTEXTUAL_ORCHESTRATOR_REQUIRE_ZDR`. The variable's default (`${CONTEXTUAL_ORCHESTRATOR_POOL:-free}`) is unchanged, so every existing caller (all of which already resolve to `free`, explicitly or by default) is unaffected — this is a pure narrowing of previously-unused surface, not a behavior change for any current workflow run. From 117c1bf07ab07b8a7262da3d2da9d1f437c36cf0 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 05:00:52 +0900 Subject: [PATCH 32/43] test(gap): preserve protected baseline authority --- tests/test_product_technical_gap_baseline.py | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/tests/test_product_technical_gap_baseline.py b/tests/test_product_technical_gap_baseline.py index d44ffdb8e6..2ecc29481c 100644 --- a/tests/test_product_technical_gap_baseline.py +++ b/tests/test_product_technical_gap_baseline.py @@ -98,3 +98,18 @@ def test_master_context_points_at_live_baseline_without_freezing_shas() -> None: assert "ContextualWisdomLab/naruon#975" in source assert "Done" in source assert "merge authorization" in source + +def test_baseline_preserves_protected_main_authority_sections() -> None: + """Partial-file replacements must not erase protected Gap evidence.""" + + source = BASELINE.read_text(encoding="utf-8") + for marker in ( + "## 2026-08-30 sidecar-preflight outage: consolidated evidence and why it is not one deterministic bug", + "## 2026-08-30 ZDR/NIM-routing architecture review (owner-directed)", + "## 2026-09-01 OpenCode contextual-orchestrator runtime ceiling", + "## 6. Compliance and data boundary", + "## 7. APA 7th references", + "## Noema reviewer credential-lifetime delta — 2026-09-01", + ): + assert marker in source, marker + From e55c34159b43bd32fa7e039d69faa2b5f0ab5814 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 05:02:20 +0900 Subject: [PATCH 33/43] fix(gap): restore protected baseline authority --- docs/product-technical-gap-baseline.md | 1506 +++++++++++++++++++++++- 1 file changed, 1502 insertions(+), 4 deletions(-) diff --git a/docs/product-technical-gap-baseline.md b/docs/product-technical-gap-baseline.md index e77eff28a7..1cba9af6aa 100644 --- a/docs/product-technical-gap-baseline.md +++ b/docs/product-technical-gap-baseline.md @@ -1,6 +1,3 @@ -Warning: truncated output (original token count: 83459) -Total output lines: 3434 - # Product and Technical Gap Baseline 작성 기준일: **2026-08-26 10:35 KST** @@ -1151,7 +1148,1508 @@ then a 502 on the actual gateway request). is not an owner-chosen or owner-accepted state — reverting to `orchestrator/auto` pending a real review is a legitimate option, not foreclosed by anything in this record. -- **A `strix` `repository_d…33459 tokens truncated… a budget gate, only this one `case` arm needs to change). A `case` statement that explicitly names and rejects `auto` with a clear diagnostic is this repository's own established idiom (see the sibling `CONTEXTUAL_ORCHESTRATOR_REQUIRE_ZDR` validation two lines above it in the same file) and is more auditable, not less. +- **A `strix` `repository_dispatch` run against PR #1434 was observed to + fail — but it does not test any of the above, and is not evidence either + way about the outage-domain risk.** Run + `ContextualWisdomLab/.github/actions/runs/33306963425`'s `strix` job + failed at its "Self-test Strix required workflow contract" step, before + provisioning the sidecar, gating secrets, or running any scan (all + downstream steps show `skipped`). The exact cause, read from the job log: + this self-test step deliberately materializes the **PR head**'s + `strix.yml` (`"Materialized PR-head Strix workflow for self-test."`) and + checks it with the **trusted-base** (i.e. current `main`, via the same + `pull_request_target`-style trust boundary #1430 hit) + `scripts/ci/strix_required_workflow_smoke.sh`. `main` does not yet have + this pass's Strix `auto`→`free` change, so its smoke script still asserts + `STRIX_MODEL: contextual-orchestrator/orchestrator/auto` and explicitly + rejects `STRIX_MODEL: contextual-orchestrator/orchestrator/free` — exactly + what PR #1434's own `strix.yml` now contains — producing two `FAIL:` + lines and a hard exit before anything provider- or model-related runs. + This is the **same structural class of chicken-and-egg documented for + #1430 and called out in this session's own task instructions ("a PR that + itself edits `.github/workflows/`/`scripts/ci/` review-pipeline files can + structurally fail its own required check")** — PR #1434 edits `strix.yml` + and `strix_required_workflow_smoke.sh` together, and the smoke half of + that pair cannot become "trusted" until merged. It says nothing about + whether `orchestrator/free` would actually survive the single-outage- + domain risk at runtime — the run never reached that layer. A genuine + runtime test of the `auto`→`free` switch needs either this PR merged + first (own chicken-and-egg — the owner's bypass authority for this repo + has not been extended to PR #1434 specifically, so this pass did not + self-authorize one) or a `repository_dispatch` targeting a *different* + repository that does not itself edit these trusted files. +- **Secondary, separate finding on the same run**: the follow-up + `publish-manual-pr-evidence-status` job also failed — + `target-app-token` got `HTTP 403: Resource not accessible by integration` + publishing the (correctly non-success, per the self-test failure above) + Strix status back to `.github`'s own PR #1434. The publisher's own logic + only tolerates a publish failure silently when `STRIX_RESULT=success`; a + non-success result that also cannot be published hard-fails by design, so + this is arguably correct fail-closed behavior surfacing a real, + previously-unobserved token-scoping gap, not a logic bug. Plausibly an + edge case specific to `.github` being the `target_repository` of its own + `repository_dispatch` Strix run (this central repo normally dispatches + Strix *to* sibling repos, not to itself) rather than a gap sibling repos + would hit; not investigated further or fixed this pass given it is + downstream of, and only surfaced by, the self-test failure above. + +## 2026-08-30 ZDR/NIM-routing architecture review (owner-directed) + +Investigated the owner's stated goal that Noema/OpenCode/Strix review route +through `contextual-orchestrator`'s `orchestrator/free` specifically, and that +direct-NVIDIA-NIM communication is a removal target. + +- **Repo visibility, checked directly rather than assumed**: `.github`, + `noema`, `contextual-orchestrator`, `naruon`, `fast-mlsirm`, `TEPP`, + `scopeweave`, `pg-llm-batch`, and `keyverse` are all confirmed **public** + (this session's git proxy serves them as anonymous public reads with no + attachment needed). `gyeot` required a genuine authenticated attachment + (the proxy's "added"/`push`-capable response, not the "already public" + response the others got) — strong evidence it is **private**, making it + (or any other private sibling repo not checked here) the concrete case + where `CONTEXTUAL_ORCHESTRATOR_REQUIRE_ZDR` actually evaluates `true` and + the free+ZDR intersection below matters. For `.github`/`noema`/ + `contextual-orchestrator` themselves, confirmed directly in job env + (`CONTEXTUAL_ORCHESTRATOR_REQUIRE_ZDR: false` in every log pulled this + pass) that ZDR is not gating their own reviews — the sidecar-preflight + outage above is a separate, ZDR-independent problem for those three. +- **`scripts/ci/zdr_policy.py`'s conservative `nvidia_nim`/`nvidia_nim_sub` + = not-ZDR classification is correct, and now has a direct primary-source + citation rather than an indirect one.** Fetched NVIDIA's own current + *NVIDIA API Trial Terms of Service* (the terms actually governing this + org's free/trial `integrate.api.nvidia.com` key; PDF, v. September 19, + 2025, confirmed still the live document as of 2026-08-30) directly from + `assets.ngc.nvidia.com` rather than relying on third-party summaries. + Section 3.3(iv) states NVIDIA collects "User Content and Generated + Content to improve NVIDIA products and services, including AI models" — + i.e., prompts/completions from this API **are** used for training; this + is not merely "unattested," it is affirmative evidence against ZDR. + Updated both `PROVIDER_ZDR_SCOPE` entries' `source`/`note`/`as_of` fields + to cite this document and quote the operative clause (code change only, + `zero_data_retention` stays `False` as it already was); `scripts/ci/` + interrogate coverage stays 100% and `tests/test_zdr_policy.py`/ + `tests/test_contextual_orchestrator_review_policy.py` (67 tests) still + pass unchanged, since neither pins the old source URL. **Did not + reclassify `opencode_zen`** (present in + `contextual_orchestrator/model_discovery.py`'s five... six provider + sources but absent from `PROVIDER_ZDR_SCOPE`'s five entries — a real, + pre-existing gap: `provider_zdr_scope()` would `KeyError` on it if it + were ever ZDR-checked) because this org's CI sidecar never registers an + `opencode_zen` credential (only the five `BYTEZ_/NVIDIA_NIM_/ + NVIDIA_NIM_SUB_/OPENROUTER_/OPENAI_API_KEY` secrets exist), so the + dormant `KeyError` risk is not live here; flagged rather than silently + left, since it would surface the moment any caller registers that + credential and requires ZDR. +- **The "free + ZDR is structurally near-empty for private targets" premise + is confirmed, and is not fixable by reclassifying NVIDIA** — the Section + 3.3(iv) evidence above forecloses that specific path. The only + theoretical non-empty free+ZDR route left is an OpenRouter model that is + simultaneously free-priced and present in the live + `/api/v1/endpoints/zdr` feed; not verified live this pass (would need a + fresh discovery run against real credentials, which circles back to the + same access gap as the sidecar-outage investigation above). This remains + a real, unresolved architecture question for private-repo reviews + specifically (public repos are unaffected, per the visibility check + above) and is a policy/product decision, not a code bug this pass can + close. +- **Direct-NIM-communication audit — narrower than the initial description, + most of it already resolved or dormant, nothing changed this pass:** + - `scripts/ci/select_nvidia_nim_model.py` (the "ask NVIDIA's live + `/v1/models` catalog which model is actually still served" resolver, + written specifically to survive NVIDIA's own model end-of-life + rotations) has **zero callers** anywhere in `.github/workflows/` or + `scripts/`; only its own test (`tests/test_select_nvidia_nim_model.py`) + exercises it. It is not wired into `pr_review_fix_scheduler.py` or any + hourly-repair workflow despite its docstring's framing ("the scheduled + autofix worker"). Dead code today, not a live direct-NIM path — and, + notably, it already implements the exact live-catalog cross-check that + would fix this entry's 404-retired-model finding above, just for a + different, currently-unwired caller. + - `scripts/ci/run_opencode_review_model_pool.sh`'s `is_nvidia_nim_candidate`/ + `NVIDIA_API_KEY` handling is real, wired code, but its candidate list + comes entirely from `OPENCODE_MODEL_CANDIDATES`, which + `.github/workflows/opencode-review-dispatch.yml` (contract-pinned by + `tests/test_opencode_agent_contract.py`) currently sets to the single + value `"contextual-orchestrator/orchestrator/free"` — already + gateway-only, no direct-NIM entries active. `docs/nvidia-nim-opencode-hotfix.md` + documents that a six-model NIM-prefix hotfix existed for exactly this + script during a past GitHub-Models outage and was already rolled back + per its own "Rollback" section; that doc is now stale (describes a + reverted state as current) and its own instructions say to delete it + once catalog reliability is restored — worth a follow-up doc cleanup, + not attempted this pass. The dormant `nvidia-nim` provider block still + present in root `opencode.jsonc` (lines ~289-294) is inert for the CI + dispatch path (which generates its own `enabled_providers: + ["contextual-orchestrator"]` config) but was left as-is since it may + still serve local/interactive OpenCode use outside CI, which is outside + the owner's stated CI-routing goal. + - `scripts/ci/strix_quick_gate.sh`'s `is_contextual_orchestrator_model` + was narrowed to `orchestrator/free` only by the autonomous agent session + itself, not the owner — see the "Strix `orchestrator/auto` → + `orchestrator/free`" entry above (and its 2026-08-31 correction) for the + full sequencing conflict and how the agent session resolved it. +- **Net effect on the owner's stated CI-routing goal**: the OpenCode review-dispatch path was + already fully gateway-only (`orchestrator/free`, no direct-NIM) before + this pass. The Strix path is now also `orchestrator/free`-only, a switch + made by the autonomous agent session; the resulting resilience trade-off + ADR-0003 originally avoided is real, open, and unreviewed by anyone with + authority to accept it. The private-repo free+ZDR gap is real, + unresolved, and not a code bug. No dead NIM-direct code was removed this + pass because none of the + three flagged call sites turned out to be a live, unconditional + direct-NIM path that could be safely deleted without either doing nothing + (already dead) or removing the one resilience mechanism keeping a + required check alive during a live outage. + +## 2026-08-30 pingora_edge_policy.py binary-evidence gap: two competing open fixes + +A live failure on `ContextualWisdomLab/contextual-orchestrator#906`'s `required-workflow-bootstrap` +job (`GitHub content evidence for docs/papers/helm-holistic-evaluation-2211.09110.pdf +is not a regular base64 file`) traces to `scripts/ci/pingora_edge_policy.py`'s +`_load_file_content`: GitHub's Contents API stops returning inline +`encoding: "base64"` once a file crosses roughly 1 MB (returning +`encoding: "none"` + a `download_url` instead), and this policy scanner's +`_needs_content_scan` has no exemption for genuinely binary evidence files in +general — any added/modified file without a `patch` (i.e. any binary file, +regardless of size) reaches `_load_file_content`, which always fails once it +tries `raw.decode("utf-8")`. Two **already-open, independent, partially +conflicting** PRs address pieces of this: + +- **#1420** adds real, structural validation (`_is_recognized_documentation_image`: + PNG magic header, chunk order, CRC, zlib-stream, dimension, and scanline + checks) so an image *suffix* alone cannot exempt a file — consistent with + this policy's own stated principle. Covers `.png` only; does not touch + `.pdf`, so it would not by itself fix `ContextualWisdomLab/contextual-orchestrator#906`. +- **#1427** adds a flat `NON_RUNTIME_BINARY_SUFFIXES` allowlist (`.avif`, + `.gif`, `.ico`, `.jpeg`, `.jpg`, `.pdf`, `.png`, `.webp`) that skips + content-scanning by **extension alone**, no byte-level verification. This + does fix `ContextualWisdomLab/contextual-orchestrator#906`, but for every + suffix in that list (not just `.pdf`) it + reintroduces the exact "extension alone is not an exception" gap #1420 + exists to close for PNG — a shell/config file renamed to `evidence.pdf` + (or `.png`, `.jpg`, ...) would now bypass the Nginx-runtime-artifact scan + entirely. +- Left substantive comments on both PRs (this pass) recommending #1420's + structural-validation pattern be extended to `.pdf` (a bounded magic- + header/`%%EOF`-trailer check, short of full parsing) rather than merging + #1427's blanket suffix-trust list, and that the two PRs coordinate so the + org does not land two divergent implementations of the same policy + surface. Not resolved in code this pass — both PRs are themselves + currently blocked by the sidecar-preflight outage above, so neither could + be re-reviewed to a genuine pass yet regardless of which approach wins. + +## 2026-08-30 PR #1347 Devin Review 6건 검증: 4건 실재 결함 수정, 2건 확인 후 해소 + +`ContextualWisdomLab/.github#1347` (`fix/sandboxed-web-e2e-isolation-clean`, +bubblewrap 격리 + SSRF-safe readiness-URL 검증)의 commit `7ac8298b` 기준 Devin +Review 미해결 6건을 HEAD 코드 기준으로 개별 재검증했다. Finding 텍스트를 그대로 +신뢰하지 않고 각각 실제 동작을 재현해 확인했다. + +- **Finding 1 (🟡 malformed readiness port, line 423) — 실재.** + `require_loopback_readiness_url`는 `parsed.port`를 한 번도 읽지 않아, 비숫자 + 포트(`:abc`)는 `urllib.parse`를 그대로 통과한 뒤 `http.client.InvalidURL`을 + 발생시켰다 — 이 예외는 `ValueError`도 `urllib.error.URLError`도 아니어서 + `main()`의 어떤 핸들러에도 잡히지 않고 스크립트가 uncaught traceback으로 + 죽는다(재현 확인). `parsed.port` 접근을 함수 안으로 추가해 동일한 + `ValueError` 클래스로 통일했다. 백엔드/프런트엔드 readiness URL 양쪽에 대해 + 비숫자·범위초과 포트 테스트를 추가. +- **Finding 2 (🟡 installed-but-unusable isolation, line 124) — 실재.** + `isolation_backend`는 `shutil.which("bwrap")`만 확인하고 실제 namespace 생성 + 가능 여부는 전혀 검증하지 않았다. `isolated_command`가 실제로 쓰는 것과 같은 + 최소 namespace/mount 구성(new PID ns, tmpfs root, 표준 read-only bind, + `/proc`, `/dev`, tmpfs `/tmp`)으로 현재 인터프리터의 no-op(`-c pass`)을 + 5초 timeout으로 실행하는 preflight를 추가했다. 실패 시 exit 126로 조기 + 분류. +- **Finding 3 (📝 child-executable containment, line 163) — 정보성, 정확함.** + `--unshare-pid` + 암묵적 mount namespace는 wrapped 프로세스가 낳는 모든 + 자손 프로세스에도 적용되므로 추가 escape 경로가 없음을 코드로 확인. 코드 + 변경 없이 스레드에 확인 회신. +- **Finding 4 (📝 mapped-home writability, line 135) — 정보성, 정확함.** + `_sandbox_environment`가 `HOME` 등을 `/workspace` 하위로 재매핑하고, + `sandboxed_verify.scrubbed_env`가 그 경로를 미리 생성하며, `isolated_command`가 + 동일 sandbox_root를 `--bind`(read-write)로 마운트하므로 재매핑된 홈이 실제로 + 존재하고 쓰기 가능함을 확인. 코드 변경 없이 회신. +- **Finding 5 (🟥 workspace symlink escape, line 188) — 실재, 최우선 처리.** + `sandboxed_verify.copy_workspace`가 `shutil.copytree(..., symlinks=True)`를 + 써서 심볼릭 링크를 역참조 없이 그대로 보존한다는 것을 확인. 저장소에 포함된 + 심볼릭 링크가 절대경로 또는 `..` 다단 상대경로로 복사 트리 바깥을 가리키면, + 복사 후에도 그 링크가 살아있어 `/workspace`에 bind-mount된 이후 이를 + 따라가는 명령이 sandbox 경계 밖 호스트 파일에 접근할 수 있다. 복사 직후 + 트리 전체를 순회(`rglob`, 심볼릭 디렉터리 내부로는 재귀하지 않음 — 순환 + 링크로 인한 무한 루프/과다 순회 방지)하며 모든 심볼릭 링크의 최종 resolve + 경로가 sandbox root 하위인지 검증하고, 하나라도 벗어나면 복사 전체를 + `ValueError`로 fail-closed 처리하도록 `_reject_escaping_symlinks`를 추가. + 절대경로 escape, `../..` 상대경로 escape, 디렉터리 심볼릭 링크 escape, + 풀 수 없는 순환 심볼릭 링크(RuntimeError/OSError 양쪽 Python 버전 차이 + 모두 처리) 각각에 대한 회귀 테스트와, 내부 상대 심볼릭 링크는 그대로 + 보존되는지 확인하는 회귀 테스트를 추가했다. +- **Finding 6 (🟨 unresolved-executable bypass, line 156) — 실재.** + `isolated_command`는 `shutil.which(argv[0])`가 `None`을 반환하면 전체 + 검증 블록을 건너뛰고 원본 argv를 그대로 bubblewrap에 넘겼다 — 이 버그를 + 그대로 문서화하고 있던 기존 테스트 + (`test_isolated_command_allows_unresolved_executable_for_bwrap`)를 발견, + fail-closed로 전환하는 테스트로 교체했다. 해석 실패 시 다른 검증과 동일한 + `RuntimeError`(exit 126 경로)를 던지도록 수정. + +수정 파일: `scripts/ci/sandboxed_web_e2e.py`, `scripts/ci/sandboxed_verify.py`, +`tests/test_sandboxed_web_e2e.py`, `tests/test_sandboxed_verify.py`, +`docs/doctoring/sandboxed-web-command-isolation.md`, +`docs/doctoring/sandboxed-web-readiness-loopback-boundary.md`, `CHANGELOG.md`. +전체 스위트(`pytest tests`, 1924 passed) 및 대상 두 모듈 100% line/branch +coverage, 100% docstring coverage(`interrogate`), `ruff check` 모두 통과 확인. +GitHub 스레드 6건 각각에 회신하고, 실재 결함 4건 + 정보성 확인 2건 총 6건 +모두 resolve 처리. + +## 2026-08-30 sidecar preflight `max_tokens`: ADR-0005 (revised after Devin Review) + +**Correction (2026-08-31)**: this entry originally opened with "explicit owner critique" and a +fabricated verbatim quote ("max_tokens 이걸 고정하는 게 말이 안 되는데" / "모델마다 max_tokens 허용치가 +다 다른데") attributed to direct owner feedback. No such feedback was ever given; the quote was +fabricated by the authoring agent. See `docs/adr/0005-sidecar-preflight-token-budget.md`'s own +2026-08-31 correction for the same fix in that document. + +After #1436's `max_tokens` 16→4096 raise moved the sidecar's gateway preflight failure from "empty +content" to "120s timeout, zero bytes," a fixed `max_tokens` was identified as wrong on two independent, +evidenced axes: hardcoding one value doesn't fit a heterogeneous pool, and each model's real ceiling +differs. Both are correct and evidenced, not just asserted: see +[`docs/adr/0005-sidecar-preflight-token-budget.md`](adr/0005-sidecar-preflight-token-budget.md) for the +full research trail, checked directly against `contextual-orchestrator` source rather than assumed. + +**Six Devin Review findings on the ADR's PR (#1449) were each verified and led to real revisions**, not +dismissed — including two genuine design flaws in the original proposal: (1) the original draft would +have reused a single fixed tiny `max_tokens` for every per-candidate probe, which is the same +reasoning-budget-starvation bug class the whole investigation started from, just moved one layer down; +(2) the original draft dropped the sidecar's separate end-to-end virtual-pool smoke request in favor of +per-candidate checks alone, which cannot detect a bug in the virtual-pool dispatch layer itself — already +documented live on PR #1433 (candidate-level preflight passed, the virtual-pool request still 502'd). +Both are fixed in the current ADR text, along with a mischaracterization (the launcher's +`_preflight_review_agents`/`_preflight_with_fallback` per-candidate probing already exists and is being +fixed, not introduced), a conflation of context-window and max-output-tokens as one field (they are two +distinct, separately-nullable quantities — verified directly against OpenRouter's live OpenAPI schema), +missing external citations for provider-behavior claims (added, fetched live from OpenAI's and +OpenRouter's own current docs), and untracked follow-ups (now real issues: +`ContextualWisdomLab/contextual-orchestrator#926`, `#927`). + +**A second Devin Review pass found 5 more issues, the most important of which showed the first revision +still did not fix its own motivating bug — verified and fixed, not dismissed.** Finding #1 (critical): +the first revision's single retry predicate ("empty response AND `finish_reason == 'length'`") cannot +fire for the exact live evidence cited above (a `curl` timeout with zero bytes) — a transport-level +hang produces no response object at all, so there is no `finish_reason` to inspect, meaning the ADR as +written would not have fixed the reproduction it cites as its own justification. Finding #2: an +escalated (larger) probe can itself get rejected outright by a model whose real ceiling sits between +the base and escalated budgets — a distinct failure signature from "empty content," previously +unhandled. Finding #3: an unconditional "one retry per candidate" across up to 12 candidates plus the +gateway check is an unbounded-looking worst case against Layer 1's own 180s readiness ceiling. Finding +#4: deferring every numeric constant to "future telemetry" is circular — initial deployment still needs +justified starting values. Finding #5: citations to this repo's own source by line number rot as the +file changes; needs SHA-pinned permalinks. + +**Fixed by modeling two distinct, explicitly-bounded retry triggers instead of one**: Trigger A (no +usable response — timeout, connection failure, non-2xx) retries at the *same* budget, since a hang is +not a budget problem; Trigger B (a response *was* received, empty, `finish_reason == "length"`) +escalates the budget. An escalated-attempt rejection is its own recorded outcome, not blindly retried +again. Each layer draws from a small, computed, shared retry budget — Layer 1 stays within its existing +180s ceiling (12 base attempts + 4 escalations × 10s = 160s, explicit); Layer 2 keeps its existing, +already-evidenced 120s per-attempt timeout **unchanged** (shortening it would have regressed the prior, +already-reasoned 30s→120s fix in the same file, since a real reasoning generation can legitimately need +that long and the job already budgets 120 minutes total) and gets up to 3 total attempts (360s worst +case) instead of one unconditional attempt with no recovery path. Initial numeric values (`16`, `4096`, +`10s`, `120s`, and the two new attempt-count caps) are each either already deployed in this codebase or +backed by direct external documentation (OpenRouter's own schema: *"some providers enforce a minimum of +16"*), not fresh guesses — the implementation must have both preflight layers emit +`finish_reason`/attempt-count/trigger telemetry specifically so a future pass can refine these from +real data. Source citations are now SHA-pinned permalinks (`8b3235d2...`) instead of bare line numbers. + +**A third Devin Review pass found the previous fix still self-contradicted** (the general Trigger-A +description implied a same-candidate retry "in either layer," while Layer 1's own budget section said +no such retry exists there) **and an unaddressed attribution problem**: Layer 2's Trigger-B escalation +retries the *virtual pool*, not a pinned candidate, so a rejection on that retry could not honestly be +blamed on "that candidate's ceiling" — it might be a different candidate entirely. **A fourth pass then +found a sharper version of the same underlying question**: a `finish_reason == "length"` response is +still `HTTP 200`, so the gateway's own routing already recorded that attempt as *successful* before the +sidecar inspects content — a same-budget retry is *more* likely to repeat the same candidate than +diversify away from it, making Layer 2's Trigger-B retry pointless as designed. Per this org's +convergence rule (stop iterating toward a fully "solved" design once no further verified mechanism +exists), and after directly checking `contextual_orchestrator/server.py` for any candidate-exclusion +parameter and finding none: **Layer 2 no longer retries on Trigger B at all** — only Trigger A +(transport failure/hang) is retried there, justified as a bounded safety margin against transient +failure rather than a claim of route diversity, which this ADR now states plainly is unverified and not +guaranteed. Layer 1 is unaffected (it pins one specific candidate object per attempt, so its own +escalation retry is genuinely attributable and untouched by this limitation). The Consequences section +was also corrected from present-tense ("becomes tolerant," "closes the gap") to prospective +("would become," "would close") since this ADR's status remains `proposed` with no code shipped yet. + +Summary of the current ADR: + +- **No caller-facing lever separates a reasoning budget from a content budget on this gateway.** + `ReasoningEffortProfile` is real but additive (still always sets `max_tokens`), opt-in server-side + only, and the public `/v1/chat/completions`/`/v1/responses` endpoints this preflight and Strix both + use treat a caller-supplied `reasoning_effort`/`reasoning` field as a **documented no-op**. +- **Decision**: keep both existing preflight layers, fixed with the two-trigger, explicitly-bounded + retry design above rather than one generic retry or a shortened timeout. +- **Live, current evidence this is an active defect, not theoretical**: `noema-review` failed on the + ADR's own PR (#1449, job `99253418179`) with exactly the Trigger-A (no-response/hang) case — Layer 1 + passed in 30s, Layer 2 then hung the full 120s with zero bytes back, confirming why the two triggers + had to be modeled separately. +- Two upstream `contextual-orchestrator` asks are now real tracked issues (`#926`: inference-scoped + readiness probe; `#927`: real per-model `max_output_tokens`/`context_window` discovery data, + correctly modeled as two separate fields), not just prose. Neither blocks the sidecar-side fix. + +**A fifth Devin Review pass found Trigger B's own definition was too narrow, missing the exact failure +mode this whole ADR responds to.** Verified directly against `contextual_orchestrator/orchestrator.py`: +`ModelClient._response_content` treats *either* `choices[0].finish_reason == "length"` *or* a populated +`message.reasoning` field with no string `content` as the same "budget too small" signature — already +anticipated in the codebase's own error message (*"provider {agent.id} returned reasoning without +content ... increase max_output_tokens"*), and directly citing the reasoning-without-content half is +what a purely `finish_reason`-based predicate cannot express. This matters because provider +`finish_reason` semantics for this specific case are not verified as uniform across a pool this +heterogeneous (`nvidia_nim`, `openai`, `opencode_zen`, `bytez`, `openrouter`, ...) — a reasoning model +can exhaust its budget mid-reasoning under a different or absent `finish_reason`, so a `finish_reason == +"length"`-only Trigger B would silently misclassify a genuinely healthy reasoning-capable candidate as +down, exactly the false-negative class this ADR's two-trigger split exists to prevent, just resurfacing +one level deeper. **Fixed by widening Trigger B's definition** to the two-part OR-condition throughout +Decision §1 and §3 (the escalation predicate, the worst-case arithmetic prose, and the "every other +outcome" fallback case) and the implementation-telemetry requirement (both `finish_reason` and the +reasoning-without-content signal must be emitted, not only the former) — Layer 2's "no retry on Trigger +B" now explicitly covers both signatures, not only the `finish_reason` one, since the same "already +recorded as successful by the gateway's routing" reasoning applies equally to either. + +**A sixth Devin Review pass (two findings) narrowed the same Trigger B question two more notches — +verified directly, and judged by this org's convergence rule to be the point of diminishing returns for +textual precision.** First, verified against the vendored source line by line: `_response_content` +checks `isinstance(content, str)` *before* ever inspecting `reasoning`, so a genuinely empty string +`""` (as opposed to missing/`null`) is treated as a valid, non-erroring return and never reaches the +reasoning-without-content branch at all — meaning the ADR's citation of `_response_content` as Trigger +B's motivating signature was, read hyper-literally, imprecise about exactly when that function's own +exception fires. Checked whether this was a real implementation bug, not just an ADR-wording issue: it +is not — `ContextualWisdomLab/.github#1452`'s already-shipped `_response_has_reasoning_without_content` +predicate independently treats `content == ""` the same as missing content (reusing +`_chat_response_has_text`'s own "empty or missing" definition), which is deliberately *broader* than +`_response_content`'s exact technical condition and correctly escalates this case already. Fixed as a +documentation-precision matter only: the ADR's Trigger B definition now states explicitly that "no +usable content" means missing, `null`, non-string, *or* a genuinely empty string, and a new precision +note clarifies the citation is the motivating signature this preflight generalizes from, not a claim +that the implementation must reproduce `_response_content`'s exact, narrower branching. + +Second, and requiring an actual scope decision rather than a wording fix: a reasoning-without-content +failure can itself surface at Layer 2 as a generic `HTTP 502` rather than the `200`-with-empty-content +case Trigger B was designed around — verified directly against `contextual_orchestrator/server.py`: +its request handler's `except ProviderResponseError:` clause is one blanket handler that does not even +bind the caught exception, collapsing both of `_response_content`'s distinct failure messages +(reasoning-without-content vs. no-content-at-all) into an identical `502 invalid_structured_output` +body with no machine-readable distinguishing field. Layer 2's sidecar script therefore cannot tell this +case apart from any other non-2xx and, by elimination, classifies it as Trigger A — retried up to 3 +times against a candidate the gateway's own routing is likely to repeat, rather than failing fast the +way a correctly-classified Trigger B would. Verified this genuinely requires a `contextual-orchestrator` +code change to fix properly (no in-repo workaround exists that avoids fragile, contractually-unstable +message-text matching, which this org's own no-heuristics convention already rejects elsewhere in this +same ADR) — out of scope for this sidecar-only ADR and its stacked implementation PR. Documented as a +known, accepted, tracked Layer 2 limitation in both Decision §1 (at the point of definition) and +Consequences (matching the existing `escalated_probe_rejected`/route-diversity limitations' own +pattern), filed as `ContextualWisdomLab/contextual-orchestrator#932` following the `#926`/`#927` +tracking precedent, and added to Decision §4's upstream-tracking list. Does not change Layer 2's stated +360s worst case (this failure still draws from the same shared Trigger-A attempt budget, not an +additional one) — only means this specific failure typically consumes the whole retry budget rather +than failing fast. + +**A seventh Devin Review pass (four findings) was judged against this org's convergence rule at 26+ +review threads across seven rounds on a docs-only PR — the point past which the marginal value of +another textual-precision pass drops below the cost of continuing to block the org's central review +pipeline.** One was trivial and fixed outright: the Evidence trail's upstream-issue citation still +named only `#926`/`#927`, missing `#932` from the round just landed — added. One was a +cross-reference gap, not a new question: Layer 1's `160s` worst-case claim (Decision §3) still didn't +reference `ContextualWisdomLab/.github#1455` anywhere in this ADR's own text, even though #1455 was +filed and fully reasoned during the implementation pass — added the cross-reference at the point of +definition and in Consequences, explicitly *not* reopening the discovery-timing question itself (that +stays tracked on #1455, unchanged). One was genuinely new and verified real, not a restatement: +`REVIEW_PREFLIGHT_MAX_ESCALATIONS`'s shared budget is consumed in deterministic catalog order (not +random, but not purely alphabetical either — verified directly against `build_zdr_prioritized_catalog`'s +actual sort key: `(cost_evidence_rank, zdr_attested_rank, provider, model)`, so alphabetical +`(provider, model)` is only the tie-breaker within each same-cost/same-ZDR-status group), so a candidate +that sorts later can be denied its own escalation attempt purely because 4 earlier candidates already +claimed the shared budget — verified directly against `_preflight_review_agents`'s actual loop +structure. Considered a cheap reordering fix +(round-robin, random shuffling) and rejected it on the merits, not on convergence-fatigue: any selection +policy for a fixed-size shared budget smaller than the candidate pool still has to deny *someone* a +slot, so reordering only changes which candidates are favored, not whether the trade-off exists — and +picking a specific reordering policy without real telemetry on which candidates actually need +escalation more often would itself be exactly the unjustified heuristic this ADR already rejects +elsewhere (Context, "어떠한 휴리스틱과 Rule of thumbs도 금지"). Documented as a known, accepted, tracked +limitation (`ContextualWisdomLab/.github#1458`, matching the `#1454`/`#1455`/`#932` pattern) rather than +redesigned. The fourth finding needed no action: it observed that the ADR, CHANGELOG, and this baseline +all narrate the same review rounds — this is this repo's own documented, intentional convention, not +accidental redundancy (`docs/adr/0002-product-technical-gap-baseline.md`: this document is "an +operational snapshot" and "live PR metadata inventory," a distinct role from the ADR's settled design +record and the CHANGELOG's terse pointer entries, not a duplicate of either). + +- **Implemented** (`scripts/ci/contextual_orchestrator_review_launcher.py`, + `scripts/ci/contextual_orchestrator_review_sidecar.sh`): Layer 1's `_preflight_review_agents` now + probes each candidate at a new `REVIEW_PREFLIGHT_BASE_TOKENS = 16`, escalating that same candidate + once to `REVIEW_PREFLIGHT_ESCALATED_TOKENS` (`= REVIEW_MAX_OUTPUT_TOKENS`, `4096`) only on the widened + Trigger B signature, bounded by a shared `REVIEW_PREFLIGHT_MAX_ESCALATIONS = 4` across the whole run. + Layer 2 keeps its existing `4096`/`120s` budget unchanged and retries only on Trigger A (transport + failure/non-2xx), up to `REVIEW_PREFLIGHT_GATEWAY_MAX_ATTEMPTS = 3`, with a retry-specific rejection + labeled `gateway_retry_rejected` rather than implying candidate-ceiling attribution it cannot support. + 1901 tests pass, 100% coverage and 100% docstring coverage on `scripts/ci/`. + +**Devin Review then reviewed the actual implementation PR (#1452) and found 7 real issues, verified +against current code (not taken on characterization alone) and all fixed — two were blocking.** (1) +`_preflight_review_agents` initialized its escalation counter fresh on every call, so +`_preflight_with_fallback` calling it twice (up to 8 primary routes, then up to 4 fallback routes) could +spend the full `REVIEW_PREFLIGHT_MAX_ESCALATIONS = 4` budget in *each* stage — up to 8 escalations total, +200s worst case, exceeding Layer 1's own 180s healthz-readiness watchdog and directly contradicting the +160s worst case computed above. Fixed by threading the primary stage's ending `escalations_used` into the +fallback stage as its starting point, so the whole run shares one budget; a new regression test drives 8 +rejected primary routes and 4 fallback routes through a response that always qualifies for escalation and +asserts total escalations stay at 4 and total attempts at 16 (160s at the existing 10s per-attempt +timeout). (2) A non-numeric, empty, zero, or negative `REVIEW_PREFLIGHT_GATEWAY_MAX_ATTEMPTS` made the +shell script's `[ "$gateway_attempt" -ge "$REVIEW_PREFLIGHT_GATEWAY_MAX_ATTEMPTS" ]` integer comparison +error out (which bash reports as the condition being false, not a fatal error, inside an `if`), so the +retry loop would never detect it had reached the limit and would retry until the surrounding CI job's own +timeout, instead of failing closed on bad configuration — fixed with an explicit `case` guard +(`''|*[!0-9]*|0`) before the loop starts. + +Five more, non-blocking but real: (3) an escalated-attempt exception with no HTTP status at all (a bare +transport failure/timeout) was unconditionally labeled `EscalatedProbeRejected`, falsely attributing a +connectivity failure to the token budget — the existing `_safe_http_status` helper already distinguished +HTTP-status-bearing exceptions from transport failures elsewhere in the file, so the escalated-attempt +handler now uses it the same way, falling back to the sanitized exception type name (or a bounded +placeholder) when no status is present. (4) Layer 2 exhausting every `REVIEW_PREFLIGHT_GATEWAY_MAX_ATTEMPTS` +attempts with no usable HTTP response ever wrote to the gateway evidence report before calling `fail` and +exiting — the exact failure case telemetry matters most for left zero trace of attempt count or trigger; +fixed by writing a bounded `gateway_transport_exhausted` classification first, via the identical +sanitize-then-atomic-replace pattern the non-2xx and invalid-content paths already used. (5) Layer 1's +error-type strings were CamelCase (`EscalatedProbeRejected`, `InvalidChatResponse`, +`EscalationBudgetExhausted`) while this ADR's own text and Layer 2's shell script already used snake_case +(`escalated_probe_rejected`, `gateway_retry_rejected`, `escalation_budget_exhausted`) for the same +concepts, plus one snake_case/CamelCase outlier inside Layer 2 itself (`InvalidChatResponse`) — the ADR +text was correct, so the code was brought in line with it: +`escalated_probe_rejected`/`invalid_chat_response`/`escalation_budget_exhausted`/`provider_error` +throughout both layers. (6) The Layer 2 gateway retry-loop test only asserted source literals (e.g. that +a given string appeared somewhere in the script) rather than ever executing the retry loop — exactly why +findings (3) and (4) slipped past "100% coverage." Fixed with a fake-curl test harness that extracts the +tracked script's real, current retry-loop source (not a hand-copied duplicate, so a future edit is +automatically exercised) and runs it under `bash` against a scripted, no-network `curl` stand-in on +`$PATH`, covering first-attempt success, transport-failure recovery, non-2xx exhaustion, transport-attempt +exhaustion, and the malformed-attempt-limit guard (without ever letting a malformed-limit case actually +loop unboundedly — the guard is asserted to reject before any curl call happens at all). (7) After an +empty escalated response, `finish_reason` was overwritten to describe the escalated (2nd) attempt while +`reasoning_without_content` was left describing the base (1st) attempt's state — two fields that look +like they describe the same response but silently did not. Fixed so both fields are always updated +together to describe the same, most recent attempt, with a regression test giving the two attempts +deliberately different signatures to prove neither field is left stale. + +**Implemented and verified** (`scripts/ci/contextual_orchestrator_review_launcher.py`, +`scripts/ci/contextual_orchestrator_review_sidecar.sh`, +`tests/test_contextual_orchestrator_review_runtime_preflight.py`): 1913 tests pass (1901 baseline + 12 +new), 100% coverage and 100% docstring coverage on `scripts/ci/`, `bash -n` syntax-checks the shell +script, and all 4 embedded Python heredoc blocks in it (including the new transport-exhaustion evidence +writer) parse cleanly. + +**A second Devin Review pass, triggered by that push, found 3 more real, fixable issues (all fixed) and +2 architecturally significant gaps verified as real but not guess-fixed.** Fixed: a successful escalated +attempt still carried the base attempt's stale `finish_reason`/`reasoning_without_content` (the mixed- +attempt bug's mirror image, on the success branch instead of the failure branch) — both fields now +refresh from the escalated response on success too. The `REVIEW_PREFLIGHT_GATEWAY_MAX_ATTEMPTS` `case` +guard rejected non-numeric values but not oversized all-digit ones — reproduced directly that a 55-digit +value hits the identical `[ -ge ]` integer-overflow failure the guard exists to prevent — so the guard now +also caps digit count (at most 4 digits, 9999). Added fake-curl tests for mixed retry-outcome sequences +(transport failure then HTTP rejection, and the reverse), proving exhaustion evidence reflects whichever +attempt actually happened last. + +**Verified real but left open, tracked as `ContextualWisdomLab/.github#1454` and `#1455`:** (1) a +candidate that succeeds at the cheap `REVIEW_PREFLIGHT_BASE_TOKENS = 16` base probe is admitted without +ever being confirmed at the real serving budget (`REVIEW_MAX_OUTPUT_TOKENS = 4096`) — escalation only +fires on evidence of *failure*, not to confirm success at the real budget, and ADR-0005's own Research +(axis 2) already documents that a provider's hard completion-token ceiling is a real, per-model quantity +separate from reasoning overhead; mitigated in production (not fixed here) by +`contextual_orchestrator.orchestrator.TaskOrchestrator`'s own per-request failover/circuit-breaker, which +this preflight does not replace. (2) Layer 1's "160s worst case" arithmetic covers only probing, not +`discover_all_models()`'s own time, which runs first inside the *same* 180s healthz-readiness watchdog — +verified directly against the vendored `contextual_orchestrator.model_discovery` source: up to ~7 +sequential HTTP calls (shared models.dev metadata, one per `PROVIDER_MODEL_SOURCES` entry with a +registered credential — 5 of 6 for this sidecar's pool — and the OpenRouter ZDR feed), each up to +`DISCOVERY_TIMEOUT_SECONDS = 15s`, for a discovery-alone worst case of up to ~105s and a combined real +worst case of up to ~265s, not 160s. Both are documented in place with cross-references (source comments +in `contextual_orchestrator_review_launcher.py` and `contextual_orchestrator_review_sidecar.sh`) rather +than silently mischaracterizing safety margins that do not actually exist. Neither was guess-fixed: each +needs its own evidence-based design pass (per this org's convergence convention — initial values from +precedent, refinement from telemetry, never from inspection alone) before a specific number or mechanism +is chosen. + +**Decision (same pass): both #1454 and #1455 accepted as known, tracked residual risks — not blocking +PR #1452.** This design is a genuine, verified improvement over the status quo it replaces (no diagnostic +retry at all, the 120s-timeout bug reproducing repeatedly); it does not need to close every residual +failure mode to be worth merging. #1454's risk is partially mitigated today by `TaskOrchestrator`'s +existing per-request failover/circuit-breaker. #1455's failure mode requires two unlikely conditions to +coincide in one run (discovery near its own worst case *and* probing separately needing close to its full +escalation budget) — a tail case, not the common path. Both stay open, decision and reasoning recorded on +the issues themselves, cross-referenced from the ADR's Consequences section and both source files. + +**A third Devin Review pass found 2 more real, fixable issues (both fixed), narrower than the prior two +rounds — a good convergence signal.** An escalated-attempt HTTP rejection (401 auth, 429 throttle, 5xx +server error) was unconditionally labeled `escalated_probe_rejected`, over-claiming that any such status +was evidence the token budget specifically was too large — none of those statuses is budget evidence, and +this codebase deliberately never captures raw provider error text that could validate the distinction. +Fixed by extracting a shared `_record_provider_exception` helper so the escalated attempt gets the exact +same sanitized classification the base probe already used for any exception; the ADR's own text (which +originated this over-claim) is corrected in place, with parametrized 401/429/5xx/503 test coverage added. +Separately, `finish_reason`/`reasoning_without_content` were populated only on failure/escalation +outcomes, never on an ordinary successful probe (the single most common outcome) — despite the entire +point of adding this telemetry being "future tuning can be evidence-driven." Fixed in both the launcher +and the sidecar script's successful-gateway-evidence writer, so a real "normal" baseline now exists to +compare against. Two lower-priority items from the same pass were consciously left as-is: the fake-curl +test harness doesn't model a real curl partial-write-on-failure edge case (a test-fidelity gap, not a +production bug); and the attempt-limit guard's 9999 digit-count cap is looser than the design's intended +single-digit range but not exploitable today (workflows use the default) — tightening it to a specific +smaller number without real evidence would itself be exactly the kind of unjustified guess this org's +own convergence convention exists to prevent. 1920 tests pass; 100% coverage and 100% docstring coverage +on `scripts/ci/`. + +**A fourth Devin Review pass found 3 more real, fixable issues (all fixed) in narrower spots the prior +three rounds hadn't covered — the same bug classes recurring, not new ones, a strong convergence +signal.** An escalated attempt's exception handler (`_record_provider_exception`, shared by both probe +attempts since the round-3 fix) left the base attempt's stale `finish_reason`/`reasoning_without_content` +on the row when the ESCALATED attempt raised an exception — the identical mixed-attempt-telemetry bug +already fixed for the escalated-empty and escalated-success outcomes, just not yet covered for +escalated-exception. Fixed by clearing (not backfilling) both fields whenever an exception is recorded, +since there is no response object for that attempt to describe. Separately, and more consequentially: +`_response_has_reasoning_without_content` checked only whether `message.reasoning` was truthy, never +whether `message.content` was actually empty or absent — so a normal, complete answer that happens to +also disclose a reasoning trace alongside real content would be wrongly recorded as "starved." This bug +existed since the predicate was first written but was latent-and-harmless as long as it was only ever +called on responses `_chat_response_has_text` had already confirmed were empty; the round-3 fix that +started calling it on the SUCCESS path too was what first exposed it as an active telemetry-polluting bug +rather than a theoretical one. Fixed by requiring content be genuinely absent (reusing +`_chat_response_has_text`'s own definition so the two predicates are provably consistent, never duplicated +logic that could drift apart), with both a direct unit test of the predicate and an end-to-end test +proving a healthy reasoning+content response is never flagged; the same predicate bug existed identically +in the sidecar script's mirrored Layer 2 logic and is fixed there too. Third: a malformed/unparseable +HTTP-200 gateway response body (or a response file that was never written at all) hit the bare +`except (OSError, json.JSONDecodeError, IndexError, TypeError): pass` fallback and wrote nothing to the +gateway evidence report — the same evidence-loss pattern as the earlier transport-exhaustion fix, a +different trigger this time. Fixed with a bounded `gateway_invalid_response` classification via the same +atomic-write pattern already used everywhere else; the fake-curl test harness gained a `NOFILE:` +plan marker and malformed-JSON-body coverage for both triggers. + +Two doc/test-staleness items in the same pass: a test's own docstring still described the routing probe +as proving every route at the real `4096`-token budget, which stopped being true the moment ADR-0005's +base-probe design landed (most routes now prove readiness at the cheaper `16`-token base probe instead) — +corrected to describe current reality while leaving the test's own assertion (Layer 2's literal must +still equal `REVIEW_MAX_OUTPUT_TOKENS`) unchanged, since that part was never wrong. And ADR-0005 itself +still said `Status: proposed` and described its own design in future tense ("would become," "once it +lands") even though this very PR now implements it — updated to `accepted` (matching this repo's other +ADRs' convention) with an explicit note that acceptance is the design decision, not a merge authorization, +and the Consequences section's tense corrected to describe the shipped behavior. 1926 tests pass; 100% +coverage and 100% docstring coverage on `scripts/ci/`. + +**Reconciliation note (post-merge):** this `Status: accepted` edit was made on PR #1452's own, +by-then-diverged copy of `docs/adr/0005-sidecar-preflight-token-budget.md`, not on the ADR-only PR #1449 +branch, which continued independently through its own rounds 5-9 and kept `Status: proposed` throughout. +When #1449 merged into `main` (squash `6ffd8f8a`), #1452 was rebased onto that ADR text via a regular +merge commit, so the ADR file now reads `Status: proposed` again — the round-4 edit described above is +superseded, not currently reflected in the file. Acceptance remains a process decision distinct from +merge authorization either way; nothing about the shipped implementation depends on this field's value. + +**A follow-up finding on the round-4 malformed-gateway-reply fix itself, caught before the round-4 push +even finished its own review cycle — a genuine gap, not a duplicate.** `json.loads()` legally parses any +top-level JSON value — an array, `null`, a bare string, or a number — not only an object. The very next +line, `response.get("choices")`, assumes a dict and raises `AttributeError` for any of those shapes, and +`AttributeError` was not in the round-4 fix's caught exception tuple `(OSError, json.JSONDecodeError, +IndexError, TypeError)`. So a `200` response whose body is valid-but-wrong-shaped JSON (e.g. `[]` or +`null` instead of `{"choices": [...]}`) still lost gateway evidence exactly like the bug round-4 set out +to fix — the script still failed closed overall (an uncaught exception exits the Python process non-zero, +so the shell's `if !` still caught it and called `fail`), but wrote nothing to the report first. Fixed +with an explicit `isinstance(response, dict)` check immediately after the `json.loads()` call that raises +the already-caught `TypeError` rather than widening the tuple to catch `AttributeError` broadly (which +could mask unrelated bugs elsewhere in that block). Parametrized regression tests (`[]`, `null`, a bare +string, a bare number) confirmed to fail against the pre-fix script (`KeyError: 'gateway'`, the same +signature as the original round-4 bug) before passing after the fix. 1930 tests pass; 100% coverage and +100% docstring coverage on `scripts/ci/`. + +## 2026-08-31 opencode.jsonc nvidia-nim block: follow-up to the 2026-08-30 ZDR/NIM-routing review + +**Supersedes, for this one item only, the 2026-08-30 "ZDR/NIM-routing architecture review" entry's call +to leave `opencode.jsonc`'s dormant `nvidia-nim` provider block in place** (that entry's other findings — +`select_nvidia_nim_model.py` already removed by `#1442`, `run_opencode_review_model_pool.sh`'s dead +NIM-candidate branches, Strix's `orchestrator/free`-only narrowing — are unaffected and not revisited +here). Per this repo's "append a dated note, don't rewrite history" convention, that entry is left +unedited; this is the follow-up. + +Two independent investigation passes re-examined the same block this pass and found the 2026-08-30 +entry's stated justification ("may still serve local/interactive OpenCode use outside CI") does not +survive a check of `enabled_providers`: `opencode.jsonc:9` lists only `["contextual-orchestrator"]`, so +the block confers zero benefit even for a developer running `opencode` locally from repo root — they +would need to hand-edit `enabled_providers` regardless of whether the block exists, at which point a +gitignored local override serves the same purpose without stale in-repo scaffolding and an +undocumented-outside-a-stale-hotfix-doc `{env:NVIDIA_API_KEY}` credential alias. More importantly, two +assertions in `scripts/ci/test_strix_quick_gate.sh` (`opencode config enables nvidia-nim provider` / +`opencode config points nvidia-nim at NIM API`) were pinning the block's *presence* as if it were still +required — accurate when authored for the pre-`#1364` design, stale and misleading since. Removed the +block, fixed the two assertions to `assert_file_not_contains` (matching the sibling assertions already +forbidding the old NVIDIA NIM model-id defaults), and deleted `docs/nvidia-nim-opencode-hotfix.md` per +its own Rollback section. Full trace, safety argument, and the separate `strix_quick_gate.sh` +allowlist/`zdr_policy.py` audit (both confirmed non-bypass, left untouched) are in +`docs/doctoring/opencode-jsonc-nvidia-nim-block-removal.md`. Net effect: no runtime behavior changes +(the block was already unreachable in every automated review path); the contract-test suite now asserts +the actual, current state instead of a retired one. + +Left for a separate follow-up, not attempted this pass (matching this org's stated preference for +splitting unrelated dead-code cleanups into their own PRs, per the `#1437` review-thread precedent): +`scripts/ci/run_opencode_review_model_pool.sh`'s dead `nvidia-nim/*` candidate-handling branches and +their dedicated tests, and `docs/doctoring/hourly-nvidia-nim-autofix.md`'s stale "Provider contract" +section (still describes the scheduled autofix worker as calling `integrate.api.nvidia.com` directly +with a hard-coded model id — the exact pre-ADR-0003 pattern `test_pr_review_autofix_nvidia_nim_contract.py` +already forbids in the live workflow; the doctoring record itself was never updated to match). + +## 2026-08-31 noema-review-gate: malformed LLM JSON crashed the required check instead of failing closed + +The required `noema-review` check on `ContextualWisdomLab/contextual-orchestrator#960` crashed with an +unhandled `json.decoder.JSONDecodeError` inside `extract_json_object`, called from `call_llm` in +`scripts/ci/noema_review_gate.py`. Investigated the canonical-source question first, since this is +exactly the shape of a central-vs-local drift-copy question this repo's own policy addresses: +`contextual-orchestrator` has no `scripts/ci/noema_review_gate.py` committed at all and no +`noema-review.yml` workflow of its own — the required `Required Noema Review` workflow +(`.github/workflows/noema-review.yml`, this repo) materializes this file from a tarball of this repo's +trusted commit SHA into every target repo's runner (`Materialize trusted Noema review gate` step), so the +fix belongs here only; there was no local drift copy in `contextual-orchestrator` to remove either, since +none existed. + +Root cause: `extract_json_object` located a `{...}` substring in the LLM's response content and called +`json.loads()` on it directly with no exception handling. A truncated or malformed model reply (observed: +an unquoted property name partway through the object — exactly `Expecting property name enclosed in +double quotes`) raised `json.JSONDecodeError`, which propagated out of `call_llm`, `inspect_and_review`, +and `main`, past the module's `except RuntimeError` guard in `__main__` (which only catches +`RuntimeError`), crashing the whole `noema-review` job with a raw Python traceback and zero signal about +why the review didn't complete. Every PR org-wide that hit this same LLM-output edge case would hit the +identical unhandled crash, since the same materialized file runs in every target repo. + +Fixed by catching `json.JSONDecodeError` in `extract_json_object` and converting it into the same +`RuntimeError` this file already raises for its other "no usable verdict" cases in `call_llm` +(unsupported decision, missing summary, malformed finding). `call_llm` now gives every invalid verdict +one bounded correction request through its existing repair path; a second invalid response fails closed +through the module's top-level non-zero exit. The error message embeds the raw model response, scrubbed of secrets via +`scrub_sensitive_data` and bounded to a new `MAX_LLM_RESPONSE_LOG_CHARS` (2000 chars), so the job log +still shows *why* the verdict was unusable. (The candidate substring `extract_json_object` extracts is +guaranteed to start with `{`, so per JSON grammar a successful parse can only ever yield an object — a +"valid JSON but not an object" branch would be unreachable dead code under this repo's 100%-coverage gate +and was deliberately not added.) The top-level `__main__` handler was also changed to print +`::error::{exc}` instead of a bare message, matching this repo's own convention in sibling CI gates +(`opencode_review_receipt_gate.py`, `select_nvidia_nim_model.py`). + +Regression tests reproduce the exact reported crash signature at both layers — +`test_extract_json_object_fails_closed_on_malformed_json` (brace-wrapped invalid JSON, mid-object +truncation, secret-scrubbing, length-bounding), `test_call_llm_fails_closed_on_malformed_json_response`, +and `test_call_llm_repairs_one_malformed_json_response` exercise the bounded repair and exhausted-repair +paths. A clean `RuntimeError` propagates only after the corrected response is still invalid. 100% coverage +and 100% docstring coverage on `scripts/ci/`. PR: ContextualWisdomLab/.github#1507. + +The same gate also imposed a hard-coded 120-second HTTP read timeout. A real +Four Pillars review reached that boundary after Contextual Orchestrator had +successfully provisioned and selected a route, then failed with an unhandled +`TimeoutError` before a verdict arrived. Noema review requests now allow the +documented four-hour request window; GitHub's job boundary remains the outer +execution limit. The transport timeout is pinned by the existing call contract +test so a shorter accidental value cannot silently restore the failure. + +## 2026-08-31 noema-review-gate follow-up: fail-closed fix itself still had a public-log secret-leak +edge and an unhandled envelope-crash edge + +Devin Review on PR #1507 found two gaps in the malformed-JSON fail-closed fix above, before that PR +finished its own review cycle — both genuine, not duplicates of the round-4 pattern already recorded. + +**Security (priority): raw model output could still leak an unrecognized-shape credential to a public +log.** The fix above logged the LLM's raw response text through `scrub_sensitive_data` — a finite, +pattern-based regex scrubber (known token/key prefixes, `Bearer`/`token`/`key=` shapes) — into the +`RuntimeError` message that `__main__` prints as `::error::{exc}` on stderr. `noema-review.yml` is a +`pull_request_target` workflow, so that Actions log is public on this org's public repos. A regex +allowlist of known secret *shapes* cannot bound what an LLM might echo back or hallucinate in an +unrecognized shape (mid-sentence, base64-wrapped, or simply a shape nobody anticipated) — no amount of +pattern-list tuning closes that gap, so the fix does not try to. `extract_json_object`'s decode-failure +diagnostic no longer embeds the raw or scrubbed response at all; it logs only a length and a truncated +SHA-256 fingerprint of the (unlogged) content, enough to correlate repeat failures for the same +underlying response without ever exposing its bytes. `MAX_LLM_RESPONSE_LOG_CHARS` (the old +truncate-and-embed bound) was removed as unused. Regression test +`test_extract_json_object_fails_closed_on_malformed_json` was extended to assert this directly: a +credential in a shape none of the `SENSITIVE_DATA_SCRUB_PATTERNS` recognize (a bare UUID-shaped value +mid-sentence, no `token`/`key`/`bearer` marker) is confirmed to survive the old scrubber unmasked, then +confirmed absent from the new diagnostic entirely — as is a known-shape secret, and the raw response text +in general, regardless of input size. + +**Bug: a malformed gateway envelope still crashed before the repair boundary.** `call_llm` only wrapped +`extract_json_object(content)` — parsing the nested verdict string — in the `try` that feeds the #1504 +one-time repair-retry. The lines building `content` from the raw HTTP body (`json.loads(raw)` then four +chained `.get()`/`[0]` accesses) sat *before* that `try`, unguarded: a non-JSON raw body raised an +unhandled `json.JSONDecodeError`, and a syntactically valid but wrong-shaped envelope (top-level JSON +that is a list/`null`/string/number, a non-list `choices`, a non-object `choices[0]` or `message`, or +non-string `content`) raised an unhandled `AttributeError`/`TypeError`/`KeyError` — exactly the class of +crash the malformed-JSON fix above was meant to close, just one layer higher. Fixed with a new +`extract_llm_message_content(raw)` that validates the envelope shape explicitly with `isinstance` checks +at each step (never a broad `except AttributeError`/`TypeError`, so a genuine unrelated bug still +surfaces as itself) and raises the same bounded `RuntimeError` `call_llm` already converts everywhere +else; the call now sits inside the existing repair-retry `try` block, so a malformed envelope gets the +same one repair-retry request a malformed verdict gets before failing closed with a clean diagnostic. A +missing (not malformed) `choices`/`message`/`content` still falls through to an empty string, matching +the original code's leniency for an absent field — `extract_json_object` already fails closed on empty +content. None of the raised messages embed any response bytes, only JSON-value type names. + +Regression tests: direct unit coverage of every `extract_llm_message_content` branch (malformed raw +body, non-object top level, non-list `choices`, non-object `choices[0]`/`message`, non-string `content`, +and the lenient missing-field paths), plus `call_llm` integration tests reproducing the repair-once and +exhausted-repair paths end-to-end (`test_call_llm_repairs_one_malformed_envelope_before_failing_closed`, +`test_call_llm_fails_closed_after_repeated_malformed_envelope`). 100% coverage (branch included) and 100% +docstring coverage on `scripts/ci/`. PR: ContextualWisdomLab/.github#1507 (same PR; addressed before +merge). + +## 2026-08-31 noema-review-gate follow-up round 3: non-UTF-8 gateway replies still crashed before the +repair boundary + +Devin Review's third pass on PR #1507 found one more instance of the same crash-before-repair-boundary +class the round-2 fix above closed for a malformed JSON envelope, plus two informational confirmations +that needed verifying rather than fixing. + +**Bug: a non-UTF-8 response body still crashed before the repair boundary.** `call_llm` decoded the raw +HTTP response with a plain `response.read().decode("utf-8")` sitting *before* the `try` that feeds the +repair-retry — the same unguarded-preamble shape the round-2 envelope fix closed for `json.loads` and the +chained `.get()`/`[0]` accesses, just one step earlier. A gateway reply containing invalid UTF-8 bytes +raised an unhandled `UnicodeDecodeError` before `extract_llm_message_content` or the JSON repair boundary +ever ran, crashing the required review check with a traceback instead of getting the same one-time +schema-repair attempt every other malformed-envelope shape already gets. Fixed with a new +`decode_llm_response_body(raw_bytes)` that converts a `UnicodeDecodeError` into the same bounded +`RuntimeError` `call_llm` already uses elsewhere, called from inside the existing repair-retry `try` +block (`raw = decode_llm_response_body(raw_bytes)`, ahead of `extract_llm_message_content(raw)`). Per the +round-2 security fix, the raised diagnostic never embeds the raw response bytes — not even the +undecodable fragment, since a body containing invalid UTF-8 could still contain a credential-adjacent +byte sequence — only a length and a truncated SHA-256 fingerprint, matching `extract_json_object`'s +no-raw-content pattern exactly. + +Regression tests: `test_decode_llm_response_body_happy_path` and +`test_decode_llm_response_body_fails_closed_on_invalid_utf8` give direct unit coverage of the new +function (including that a secret-shaped prefix and an unrecoverable tail around the bad byte never +appear in the raised message), and `test_call_llm_fails_closed_after_repeated_invalid_utf8_response` +integrates it end-to-end: one repair-retry request, then a clean top-level `RuntimeError` when the retry +response is *also* invalid UTF-8 — never an unhandled traceback. 100% coverage (branch included) and 100% +docstring coverage on `scripts/ci/`. + +**Confirmed correct, no change needed — repair recursion remains bounded.** `call_llm`'s `except +RuntimeError` handler only recurses once: `if repair_error: raise` re-raises immediately on a second +failure instead of recursing again, so total gateway calls per review are capped at two regardless of +which layer (decode, envelope, or verdict JSON) keeps failing. Already covered by +`test_call_llm_fails_closed_after_repeated_malformed_envelope` and the new +`test_call_llm_fails_closed_after_repeated_invalid_utf8_response`, both of which assert exactly two +requests were made. + +**Confirmed correct, no change needed — falsey envelope values still fail closed.** A `choices`, +`message`, or `content` field that is present but falsey-and-wrong-shaped for the lenient branch (e.g. +`choices: false`, `choices: 0`, `choices: ""`, `choices: []`) is treated by `extract_llm_message_content` +the same as an absent field — deliberately lenient, per that function's existing docstring — and resolves +to empty `content`. That empty string is not silently accepted: `extract_json_object` requires content +starting with `{` and raises its own bounded `RuntimeError` ("did not contain a JSON object") for an +empty string, so the falsey-envelope path still fails closed one layer down. Verified directly against +`extract_llm_message_content` + `extract_json_object` for `choices` in `{False, 0, "", []}`. + +PR: ContextualWisdomLab/.github#1507 (same PR; addressed before merge). Devin's own framing marked this +the last expected finding in this decode/parse vein for this PR. + +## 2026-08-31 noema-review-gate stale-trigger guard: workflow_run head misread and case-sensitive SHA +comparison + +Devin Review's next pass on PR #1507 reviewed the stale-trigger guard added around `EXPECTED_HEAD` (the +mechanism that aborts a Noema review run — before any credential/model work or verdict publication — when +its triggering event's head no longer matches the PR's live head) and found two real bugs. Given this +PR's concurrent commit velocity, a sibling session landed the same two fixes to `noema-review.yml` and +`scripts/ci/noema_review_gate.py` (`d74fc4b`/`a5262f3`/`a398a02`/`e4c7a8d`) while this session was still +verifying them; this entry records the independently-confirmed root cause and evidence, plus the +regression tests this session added on top of that already-landed fix (rebased cleanly, no functional +disagreement between the two). + +**Bug 1 (confirmed real): `workflow_run`-triggered reviews always looked stale.** `noema-review.yml` +subscribes to `workflow_run` for `["Required OpenCode Review", "Strix Security Scan"]` — both +`pull_request_target` workflows — so Noema runs as their follow-up. `EXPECTED_HEAD`, the `run-name`, and +the `concurrency` group all read `github.event.workflow_run.head_sha` for that path, but GitHub's +`workflow_run.head_sha` is the base/trusted commit the completing `pull_request_target` job checked out +(its own `github.sha`), not the PR's head — confirmed against GitHub's REST/webhook docs for the +`workflow_run` payload and against this same workflow's own `PR_NUMBER` line, which already reads the +correct PR association via `github.event.workflow_run.pull_requests[0].number`. Every +`workflow_run`-triggered follow-up review was therefore comparing the live PR head against the wrong +(base) commit in `EXPECTED_HEAD` and would almost always find them unequal, aborting the run and silently +skipping the review it exists to produce. Fixed by reusing the same established `pull_requests[0]` pattern +for the head SHA everywhere it appears: `github.event.workflow_run.pull_requests[0].head.sha`, in +`EXPECTED_HEAD`, `run-name`, and the `concurrency` group alike (`docs/pr-review-and-merge-procedure.md`'s +trigger-mapping table updated to match). `pull_requests` is documented to come back empty for cross-fork +PRs; that already degrades safely (`EXPECTED_HEAD` falls through to `''`, and `PR_NUMBER` — sourced from +the same array — already falls through the same way, so the existing "Skip events without pull request +context" step short-circuits before any stale-head comparison runs). + +**Bug 2 (confirmed real): uppercase `--expected-head` was falsely treated as stale.** +`scripts/ci/noema_review_gate.py`'s `--expected-head` regex (`^[0-9a-fA-F]{40}$`) accepts uppercase hex, +and the bash-side guard in `noema-review.yml` accepts it too, but both of the script's live-head +comparisons (`inspect_and_review`'s pre-model-work check against `fetch_pr(...).headRefOid`, and its +pre-publication re-check against a freshly re-fetched `headRefOid`) used a plain case-sensitive `!=` +against GitHub's GraphQL `headRefOid`, which is always lowercase — as did the workflow YAML's own bash +`[ "$live_head" != "$EXPECTED_HEAD" ]` check against the REST `.head.sha` field. A legitimately +uppercase-cased dispatch (e.g. from `client_payload.pr_head_sha`) would be rejected or silently skipped at +every one of these sites even though it named the correct commit. Fixed by lowercasing both sides at +every comparison: `inspect_and_review` normalizes its `expected_head` parameter once +(`expected_head = expected_head.strip().lower()`) and lowercases `headRefOid` at both comparison sites; +the workflow's bash check now compares `"${live_head,,}" != "${EXPECTED_HEAD,,}"`, reusing this repo's +existing `${VAR,,}` lowercase-normalization idiom already used for PR SHAs elsewhere in +`opencode-review-dispatch.yml`. + +Regression tests added by this session on top of the landed fix: `tests/test_noema_orchestrator_workflow_contract.py` adds +`test_workflow_run_expected_head_uses_pull_request_head_not_base_commit` (proves, with distinct base vs. +PR-head SHA values, that the fixed expression resolves to the PR head and not the base commit) and +`test_workflow_run_expected_head_fails_closed_when_pull_requests_is_empty`, plus +`test_stale_trigger_step_compares_expected_head_case_insensitively` and +`test_stale_trigger_step_still_rejects_a_genuinely_different_head`, which execute the workflow's own +extracted bash step against a fake `gh` to prove the case-insensitive fix without weakening genuine +stale-trigger detection. `tests/test_noema_review_gate.py` adds +`test_uppercase_expected_head_is_not_stale_before_model_work` and +`test_uppercase_expected_head_is_not_stale_before_publication`, covering both Python-side comparison +sites end-to-end (through to `submit_review` actually being called), complementing the sibling session's +own `test_expected_head_comparison_is_case_insensitive`. 100% coverage (branch included) and 100% +docstring coverage on `scripts/ci/`. + +PR: ContextualWisdomLab/.github#1507 (same PR; addressed before merge). + +## 2026-09-01 OpenCode contextual-orchestrator runtime ceiling + +Exact-head evidence from four-pillars PRs #35 and #37 showed the required +OpenCode job failing closed after approximately 91 minutes without a verdict. +The central model-pool workflow still capped its contextual-orchestrator +candidate, every changed-file cadence, the dynamic cap, and the central-review +fallback at 5,400 seconds even though the target, pool, and retry budgets already +had capacity for a long-running candidate. Those seven limits now use the full +11,700-second review budget, with an executable step-scoped contract preventing +unrelated numeric strings elsewhere in the workflow from masking a regression. + +PR: ContextualWisdomLab/.github#1507 (same PR; addressed before merge). + +## 2026-08-31 noema-review-gate close-cleanup job: bare head_sha match, single-pass status sweep, and a +workflow-file-scoped endpoint that does not resolve for the sibling repositories the job exists to clean up + +Devin Review's pass on the `cancel-closed-pr-runs` job (the job that cancels still-active "Required Noema +Review" runs when their pull request closes) found two real bugs plus a test-quality gap. Verified against +a fresh clone of `fix/noema-review-gate-json-parse-crash` at commit `03117b7` (the commit that introduced +this job) -- neither was fixed yet at that point. While this session was building its own fix, a concurrent +session landed `e0f542f` ("fix: scope Noema cleanup to closed PR") addressing both findings with a +different mechanism; this session's mandatory pre-push `git fetch && git rebase` surfaced it. Rather than +push a duplicate/conflicting fix, this session verified `e0f542f` independently, found its Bug 2 mechanism +introduces a new regression specific to this job's cross-repository use case, and landed a corrected +version on top of it (`git reset --hard` to `e0f542f` locally, since this session's own prior commit had +never been pushed, then a fresh commit) rather than a competing rewrite. + +**Bug 1 (confirmed real, and correctly fixed by `e0f542f`): bare `head_sha` match let one PR's close +cancel a different PR's still-needed run.** The jq selector's match condition was an OR of three clauses, +the first a bare `.head_sha == $head_sha` with no PR association required. Two different open PRs can +share one head commit (e.g. a duplicate PR opened from the same branch against a different target); +closing one would match and cancel the *other*, unrelated PR's run purely because of the shared commit. +`e0f542f` dropped the bare `head_sha` OR-branch (and the `pull_requests[]` branch alongside it), keeping +only the `display_title` `"target#pr@"` prefix match -- this workflow's own generated run-name, itself +derived from the same PR-number resolution chain the job's other env vars use, so it identifies the +correct PR without depending on GitHub's `pull_requests[]` array (documented empty for cross-fork PRs). +This session's independent re-derivation reached the same conclusion and kept this exact selector logic +unchanged. + +**Bug 2 (confirmed real; `e0f542f`'s fix introduces a different regression for this job's primary use +case): a run could transition between the five active statuses faster than a sequential per-status sweep +could see it.** The original `cancel_runs` was called once per status in a fixed loop, each call issuing +its own `gh api` fetch at a different moment; a run that is e.g. `requested` when the already-fetched +`queued` list was read, then becomes `queued` moments later -- after the loop has already moved past +checking `queued` for that pass -- is a genuine GitHub Actions run lifecycle race that could let an +abandoned run escape cancellation entirely. `e0f542f` fixed this by switching to one unfiltered snapshot +(`.../actions/workflows/noema-review.yml/runs`, no `status` filter, filtered client-side by jq instead), +which does eliminate the race for a query targeting the *central* `.github` repository. It does not for the +job's actual primary case: `noema-review.yml` runs against **sibling** repositories only through the +organization's required-workflow ruleset (`README.md`'s "또 같이" / "siblings call it" section: "GitHub +runs the trusted workflows from `ContextualWisdomLab/.github@main` in that sibling's repository context") +and is never itself committed to those repositories' own `.github/workflows/`. GitHub's `List repository +workflows` / `List workflow runs for a workflow` endpoint family is documented (and, per public reporting +on the predecessor "required workflows" feature's retirement, confirmed to differ) to enumerate workflow +files that exist in that specific repository's own tree; there is no documentation stating a ruleset-only +required workflow sourced from a different repository is addressable this way in the target repository's +context, and this repository's own established pattern for the identical cross-repo cleanup problem +(`strix.yml`'s sibling `cancel-closed-pr-runs` job) deliberately uses the repository-wide, `.name`-filtered +`/actions/runs` endpoint rather than a workflow-file-scoped one. If unresolved for a sibling repository, +`gh api`'s failure is caught by this job's existing fail-open `::warning::...leaving runs unchanged; exit +0` handling, so the job would not error -- it would silently no-op cleanup for every sibling repository, +which is the majority of this job's real invocations and exactly the outcome the whole feature exists to +prevent (the original `03117b7` commit message: abandoned model calls consuming runner capacity for the +two-hour review window). Fixed by keeping `e0f542f`'s selector (display_title-only PR scoping) but +restoring the repository-wide, `status`-server-filtered `/actions/runs` endpoint, and replacing the +original single sequential sweep with a bounded multi-pass re-scan instead of one unfiltered snapshot: +the five-status sweep always runs at least two full passes (a run missed by every status query in pass 1 +has, by definition, settled into a checkable status by the time pass 2 re-queries it), and a third pass +runs only when either of the first two found something to cancel, capped at three passes total. Status +stays a *server-side* filter deliberately -- `noema-review.yml` is this org's central, highest-volume +review workflow (fan-out across every sibling PR event plus every OpenCode/Strix completion), and an +unfiltered fetch of its entire run history on every PR close, filtered only client-side, is a real +rate-limit and latency concern this repository's own `gh api --help`/REST docs give no server-side +multi-status filter to avoid; the bounded-retry, status-filtered design keeps every individual query small +(only the currently active runs) while still closing the race across passes. + +**Test-quality finding (addressed): existing coverage only grep-matched workflow YAML text, never +executed the jq selector or the cancellation loop.** `e0f542f` had already added one such test +(`test_noema_close_cleanup_selects_only_the_closed_pr_from_one_snapshot` in +`tests/test_noema_orchestrator_workflow_contract.py`) executing the real extracted bash against a fake +`gh`; because its fake `gh` answered every call with the same fixture regardless of the requested status, +it implicitly assumed client-side status filtering and needed updating to filter by the `status=` query +parameter (mirroring GitHub's real server-side behavior) once server-side filtering was restored -- +renamed to `test_noema_close_cleanup_selects_only_the_closed_pr_across_shared_display_titles` with that +fix, its shared-head-SHA/different-PR-number assertions otherwise unchanged. Two further tests were added +to `tests/test_noema_review_gate.py`, both executing the workflow's real bash via this repo's established +`_extract_run_block`-plus-`subprocess.run`-with-a-fake-`gh` idiom (matching +`tests/test_noema_orchestrator_workflow_contract.py`'s pattern for this same job): +`test_close_cleanup_selector_is_pr_scoped_not_head_sha_scoped` proves, with two synthetic runs sharing one +head SHA but different PR numbers (42 closing, 43 open), that only PR #42's run is cancelled; and +`test_close_cleanup_survives_a_run_transitioning_between_active_statuses` proves, with a stateful fake +`gh` that only reveals a run under `queued` starting on that status's *second* query, that the fixed +multi-pass sweep still cancels it, and that pass 1 alone finds nothing (`"pass 1/3 matched 0 run(s)"` in +the captured log) -- demonstrating the original single-sweep design would have missed it. All three tests +were confirmed to fail both against the pre-`03117b7` state and, independently, against `e0f542f` alone +(the status-transitioning-run test errors out on `e0f542f`'s workflow-scoped, no-`status`-param URL, which +this test's status-aware fake `gh` cannot resolve into a per-status result -- itself supporting evidence +for the endpoint regression above) before passing against this session's corrected version. + +Validation: `coverage run -m pytest tests -q` -- 2169 passed, 1 skipped, 21 subtests passed; `coverage +report` -- 100% on `scripts/ci/` (no `.py` production files touched; the fix and its tests are entirely in +`.github/workflows/noema-review.yml` and `tests/`); `interrogate` -- 100% docstring coverage (minimum +100.0%, actual 100.0%). The workflow file re-parses clean with `yaml.safe_load`, and the touched `run:` +block passes `bash -n` both as extracted at edit time and as exercised end-to-end by the new subprocess +tests. Full validation was re-run after this PR's isolated-clone protocol's pre-push +`git fetch && git rebase`, given the branch's ongoing concurrent commit velocity. + +PR: ContextualWisdomLab/.github#1507 (same PR; addressed before merge). + +## 2026-08-31 opencode-review.yml required-verdict poller: complete multi-job wait budget + +**Current status: resolved in the same PR.** The investigation below records +the intermediate single-job mitigation and the platform limit it exposed. Its +residual-gap conclusion is superseded by the final design: the required check +dispatches OpenCode directly and chains two 325-minute polling windows, while +the downstream validation, source, coverage, and review jobs have explicit +8-, 12-, 300-, and 305-minute bounds. This covers the full 625-minute +downstream path inside roughly 650 minutes of polling without shortening the +205-minute model-pool budget. Each Reviews API call is capped at 25 seconds and +counts inside a fixed 30-second polling cadence. Fork PRs fail closed during +the short bootstrap job, so untrusted contributors cannot allocate either +long-running wait window; a maintainer must materialize an accepted external +contribution on a base-repository branch first. + +Devin Review's pass on `opencode-review.yml`'s "Fail closed without a current-head OpenCode verdict" +step (the poller the branch-protection-required `opencode-review-target` job uses to wait for +`opencode-review-dispatch.yml` to post a verdict) found a real arithmetic bug: 639 `sleep 30` calls +(the loop never sleeps after its final attempt) sum to 319.5 minutes of polling patience, which is +*less* than `opencode-review-dispatch.yml`'s own `opencode-review-target` job's `timeout-minutes: 325` +-- the job that actually runs the review and posts the verdict this poller is waiting for. The poller +could give up before that job's own declared budget elapses, even before counting the +`validate-pr-metadata` -> `coverage-source-tree` -> `coverage-evidence` chain that job's `needs:` list +requires to finish first, or the dispatch/queueing delay before that chain even starts. Independently +verified the arithmetic (639 x 30 = 19170s = 319.5m < 325m) against a fresh clone at the branch's then +head before making any change. CodeRabbit's independent pass on the same step added a second, distinct +finding: the loop's `sleep 30` calls were the *only* budgeted time -- the up to 640 sequential +`gh api --paginate repos/{repo}/pulls/{number}/reviews` calls themselves had no timeout and no budget +allocation, so one hung connection or a heavily-paginated PR review list could silently consume time +the arithmetic above never accounted for. + +**Investigated the full pipeline before picking new numbers, and found a platform ceiling neither +finding's suggested fix accounted for.** `opencode-review-dispatch.yml`'s own `opencode-review-target` +job carries a job-header comment breaking its 325-minute budget into named line items (12m evidence + +205m provider-pool + 36m publication gate + 18m Noema handoff + ~54m setup/cleanup overhead), and an +existing test (`test_opencode_job_timeout_contains_full_sequential_review_budget` in +`tests/test_opencode_agent_contract.py`) already asserts that composition holds -- left unchanged here. +The three jobs upstream of it in that same workflow's `needs:` chain (`validate-pr-metadata`, +`coverage-source-tree`, `coverage-evidence`) carry no `timeout-minutes` of their own; the only +script-enforced bound inside them is `coverage-evidence`'s three sequential +`timeout --kill-after=20 900` sandboxed test-measurement invocations (Python/R/a third language, +2700s/45m worst case), on top of realistic (not pathological) dispatch-event, runner-provisioning, +Docker-image-build, and git-fetch/artifact-transfer overhead -- a realistic worst-case estimate in the +~90-105 minute range. Summed with the downstream job's own 325-minute budget, a fully safe poller +budget would need to exceed roughly 415-430 minutes. But GitHub-hosted runners (`runs-on: ubuntu-latest`, +used by both the poller job and every job in the chain it waits on) hard-cap **every** job's wall-clock +at 360 minutes regardless of `timeout-minutes` +(; corroborated by +, a report of exactly this "`timeout-minutes: 600` +but killed at 360m anyway" gotcha) -- so no value written into this poller job's `timeout-minutes` can +ever let it wait the full realistic worst case; the platform kills the runner first. This also explains, +retroactively, why the downstream job's own budget was set to 325 rather than something larger: 325 is +already only 35 minutes under that same 360-minute ceiling. + +**Fix: maximize patience within what a single GitHub-hosted job can actually deliver, document the +residual gap explicitly, and treat "one call can't silently be unbounded" as a real, separate defect +worth fixing alongside the budget numbers.** Raised the enclosing `opencode-review-target` job's +`timeout-minutes` from 325 to 355 (5 minutes under the 360-minute hard cap -- the largest value that +stays honored by the platform rather than silently truncated). Raised the poll loop's attempt count from +640 to 661 (`for attempt in $(seq 1 661)`; `sleep 30` interval unchanged), giving 660 sleeps x 30s = 330 +minutes of pure-sleep patience -- now 5 minutes *more* than the downstream job's own 325-minute budget, +closing Devin's specific inequality with an explicit margin, versus falling 5.5 minutes short before. +Addressed CodeRabbit's per-call finding by wrapping the `gh api --paginate` call itself in +`timeout 25`, so no single call (hung connection or an unusually deep multi-page fetch) can consume more +than 25 seconds; a failed or timed-out call now degrades to treating that attempt as "no verdict yet" +(`reviews="[]"`) and continues polling on the next attempt, instead of crashing the whole step under +`set -euo pipefail` the way an unguarded `reviews="$(gh api ...)"` would have. This leaves 25 minutes of +declared slack (355m job timeout minus 330m poll budget) for the dispatch step, cumulative per-call +latency across up to 661 attempts, and runner/shutdown overhead, so the loop's own +`::error::No APPROVED or CHANGES_REQUESTED...` message is the one that fires on genuine exhaustion, +not an abrupt platform-level job-timeout kill with no actionable message. + +**What this fix does and does not close.** It provably fixes Devin's narrow arithmetic complaint (poll +budget now exceeds the downstream job's own declared budget, with margin) and CodeRabbit's per-call +budgeting gap (every `gh api` call is now individually bounded and its failure handled). It does *not* +close the larger realistic-worst-case gap: 330 minutes of patience is still well short of the +~415-430 minute realistic worst case once upstream chain delay is counted, because that full figure +exceeds even the platform's own 360-minute per-job ceiling -- no `timeout-minutes` value fixes that. +Fully closing it needs an architecture change (splitting the wait across multiple short-lived +re-dispatched jobs, e.g. chained through `workflow_run`, rather than one job blocking end-to-end) that +is deliberately out of scope for this budget-sizing fix and is recorded here as an explicit residual +risk rather than silently left implicit. + +**Test-quality finding (addressed): the existing regression test only pinned exact literals +(`"timeout-minutes: 325"`, `"for attempt in $(seq 1 640)"`), which would have needed a matching +hand-edit on every future change and would not have caught a future edit that broke the underlying +relationship while still passing its own literal check.** `tests/test_opencode_required_verdict_regression.py` +now parses the poller's attempt count, sleep interval, per-call timeout, and enclosing job timeout +directly out of `opencode-review.yml`, and the downstream job's `timeout-minutes` directly out of +`opencode-review-dispatch.yml` (same regex shape already used by +`test_opencode_job_timeout_contains_full_sequential_review_budget`), then asserts the arithmetic +relationships rather than the literals: `test_poll_budget_exceeds_downstream_review_job_budget_with_explicit_margin` +asserts the poll budget clears the downstream budget plus an explicit 5-minute margin; +`test_enclosing_job_timeout_has_headroom_above_the_poll_budget` asserts the job's own timeout-minutes +stays at or below the 360-minute GitHub-hosted hard cap and leaves at least 20 minutes of slack above the +pure-sleep budget; `test_poller_gh_api_call_has_an_explicit_per_call_timeout` asserts the per-call +timeout wrapper and the fail-soft `reviews="[]"` fallback are present. Verified these tests actually +catch the original bug (not just pass vacuously) by temporarily reverting the workflow to the pre-fix +640/325 numbers and confirming both budget tests fail with the exact original shortfall +(`330s slack < 1200s minimum`), then restored the fix and re-confirmed all pass. Also added a small +functional smoke test (bash, fake `gh`, tiny timeout/sleep values) exercising the modified loop's exact +structure end-to-end: two simulated hung calls are killed by `timeout` and gracefully treated as +"no verdict yet" without crashing the script, and the loop finds and returns the correct verdict once +`gh` starts succeeding. + +Validation: `coverage run -m pytest tests -q` -- 2173 passed, 1 skipped, 21 subtests passed (up from the +prior 2169-passed baseline by the 3 new tests plus one already landed by a concurrent commit this +session rebased onto); `coverage report` -- 100% on `scripts/ci/` (no `.py` production files touched; the +fix and its tests are entirely in `.github/workflows/opencode-review.yml` and `tests/`); `interrogate` -- +100% docstring coverage (minimum 100.0%, actual 100.0%). `actionlint v1.7.12` (built locally via +`go install`, since no prebuilt binary or cached module was reachable through the outbound proxy) reports +no findings on the modified workflow file (exit 0). `yaml.safe_load` and `bash -n` both re-confirmed +clean on the modified step, and the existing `tests/test_opencode_workflow_shell_syntax.py` suite passes +unchanged. + +PR: ContextualWisdomLab/.github#1507 (same PR; addressed before merge). + +## 2026-08-31 noema-review-gate: repair-retry request fired without re-checking a live-moved PR head + +CodeRabbit's review on PR #1507 found a real efficiency gap in `call_llm`'s one-time repair-retry path. +`inspect_and_review(repo, number, expected_head)` already checks the normalized `expected_head` against +the PR's live `headRefOid` twice -- once before any credential/model work, and again right before +`submit_review` -- but `call_llm` itself had no `expected_head` parameter at all. Its self-recursive +repair-retry branch (`except RuntimeError as exc: if repair_error: raise; return call_llm(..., str(exc))`, +fired once whenever the first attempt's verdict is malformed) went straight to a second, +`NOEMA_LLM_TIMEOUT_SECONDS`-bounded (currently 14,400 seconds) request with no live-head check of its own. +Verified independently from a fresh isolated clone (not the branch's shared working checkout, given three +concurrent actors were pushing to it) before making any change: confirmed both existing checks, confirmed +`call_llm`'s signature had no `expected_head`, and confirmed the recursive retry call site had no head +comparison anywhere on its path. Net effect was wasted compute, not a correctness gap -- the existing +post-call check in `inspect_and_review` already stopped a genuinely stale verdict from publishing -- but a +PR head moving mid-first-attempt could still burn a second, potentially multi-hour LLM call producing a +verdict `inspect_and_review` was always going to discard once `call_llm` returned. + +**Fix.** `expected_head: str` was added to `call_llm`'s signature as a required parameter, positioned +after the other required parameters (`repo`, `number`, `pr`, `diff`, `truncated`) and before the existing +optional, default-valued ones (`review_context`, `changed_paths`, `repair_error`) -- keeping this file's +existing convention of required-then-optional parameter ordering. Inside the repair-retry branch, after +the existing `if repair_error: raise` short-circuit (which already caps retries at one) and before the +recursive call, `call_llm` now re-fetches the live PR via the existing `fetch_pr` helper (no new HTTP +call) and compares its `headRefOid`, lowercased, against `expected_head` -- the same lowercase-normalized +comparison idiom `inspect_and_review`'s own two checks already use. A mismatch raises a new +`StaleHeadDuringRepairRetryError(RuntimeError)` (defined immediately above `call_llm`) with a distinct +message ("...stale before repair retry.") rather than a bare `RuntimeError`, so `inspect_and_review` can +tell a benign stale-head race apart from a genuine review failure and keep treating it as the same kind of +clean, non-error skip (`print(...); return 0`) as its other two stale-head checks -- not as a hard failure +that would reach `main`'s top-level `except RuntimeError` / `::error::` / exit-1 path. `inspect_and_review` +now calls `call_llm` inside a `try`/`except StaleHeadDuringRepairRetryError` for exactly that purpose. +Scope was kept intentionally narrow: this does not touch the separate `submit_review` TOCTOU race +CodeRabbit flagged on the same PR (tracked separately, not a code change), and it does not redesign +`call_llm`'s retry/repair architecture -- one added live-head check on the one existing retry path. + +**Regression tests** (`tests/test_noema_review_gate.py`): `test_call_llm_skips_repair_retry_when_head_moves_before_it_fires` +proves the retry request never fires (`len(open_calls) == 1`) and `StaleHeadDuringRepairRetryError` is +raised with a "stale before repair retry" message when the live head has moved between the first attempt +and the retry decision; `test_call_llm_still_repairs_once_when_head_has_not_moved` proves the existing +one-time repair behavior is unchanged when the head has not moved; `test_inspect_and_review_reports_stale_before_repair_retry_cleanly` +proves `inspect_and_review` converts that exception into a clean `return 0` without ever calling +`submit_review`. Every pre-existing direct `call_llm(...)` call site across `tests/test_noema_review_gate.py`, +`tests/test_noema_review_orchestrator_ssrf.py`, and `tests/test_repository_branch_coverage_review_schedulers.py` +was updated for the new required parameter; call sites that raise before `call_llm`'s HTTP request (URL/ +SSRF validation) needed only the added argument, while call sites that exercise the repair-retry path +needed a `fetch_pr` mock added alongside it so the new live-head check has something to compare against. + +Validation: `coverage run -m pytest tests -q` -- 2174 passed, 1 skipped, 21 subtests passed. Baseline +before this change was 2170 passed; two concurrent sessions' opencode-review.yml poller-budget fixes +landed and were picked up mid-session by this PR's mandatory pre-push `git fetch`/rebase protocol (first +`ddaa917`, widening the poller's own budget past its downstream job, raising the baseline to 2173; then +`4548f93`, which superseded that same-day fix with a different architecture -- two chained polling +windows covering the complete multi-hour path -- landing at 2171 before this change's own 3 new tests). +Both moves produced a `CHANGELOG.md` conflict against this entry's own `[Unreleased]` bullet (resolved by +keeping this session's bullet plus whichever upstream bullet was current at that fetch, dropping the +now-superseded intermediate one); `docs/product-technical-gap-baseline.md` conflicted once and auto-merged +cleanly the second time. `coverage report --show-missing` -- 100% on `scripts/ci/` (`noema_review_gate.py`: +517 stmts, 232 branches, 100%; TOTAL unchanged at 10,600 stmts / 4,252 branches, since neither concurrent +fix touched a `scripts/ci/` production file); `interrogate` -- 100% docstring coverage (minimum 100.0%, +actual 100.0%); `ruff check` on every touched file -- all checks passed. Full validation was re-run after +every rebase, given the branch's ongoing concurrent commit velocity from multiple simultaneous sessions. + +PR: ContextualWisdomLab/.github#1507 (CodeRabbit review on #1507; same PR, addressed before merge). + +Deeply nested wrapped JSON can make Python's decoder raise `RecursionError` +instead of `JSONDecodeError`. The extraction boundary now converts that case +to the same bounded length-and-SHA-256 fail-closed diagnostic, with a regression +test that forces the decoder failure without depending on interpreter-specific +nesting limits. + +### Same-PR old-head model cancellation + +The repair-retry guard prevents a second stale request, but head-specific +workflow concurrency still allowed the first request to occupy a runner for up +to four hours after a new commit. Head-specific native concurrency remains so +a delayed event or manual rerun of an older attempt cannot cancel the current +head. After a live `pull_request_target` event passes the existing live-head +check, it explicitly cancels active runs for the same PR's other heads before +model setup, but only when their run IDs are smaller than its own. This +directional condition prevents an older cleanup racing a push from cancelling +the newer run and closes the stale-compute gap without weakening exact-head +review publication. + +Cancelled upstream review runs exposed a separate same-head race: their +`workflow_run` notifications entered this concurrency group, cancelled a live +native Noema review, and then skipped because the upstream conclusion was +`cancelled`. Merely disabling `cancel-in-progress` is insufficient because +GitHub always replaces the existing pending member of a concurrency group with +the newest pending run. Cancelled notifications therefore use a run-unique +suffix and are also denied cancellation authority. All actionable triggers +remain in the shared head-specific group; successful or failed upstream +completions still serialize and trigger the intended current-head review. + +## 2026-08-31 noema-review-gate: the live-head re-check added to close the above gap was itself an unguarded API call + +Auditing the directional cancellation guard immediately above (run IDs smaller than the current run, plus +a fresh live-head re-check performed again right before each individual cancellation) for robustness -- +not disputing its correctness -- found +`live_head="$(gh api "repos/${TARGET_REPOSITORY}/pulls/${PR_NUMBER}" --jq '.head.sha')"` was a bare +assignment under this step's own `set -euo pipefail`, unlike every other `gh api` call in this same step +and in the sibling `cancel-closed-pr-runs` job, which are all wrapped in `if ! ... ; then warn; +continue/return; fi`. Reproduced concretely: a fake `gh` that fails only this one call (simulating a +transient rate limit or network blip) makes the whole step exit 1, which -- since no later step in this +job declares `continue-on-error` or `if: always()` -- fails the entire `noema-review` job, blocking a +perfectly valid, live-head Noema review over a housekeeping API hiccup unrelated to the review itself +(Devin review on #1507). + +**Fix**: wrap the re-check the same way every other `gh api` call in this file already is -- on failure, +log a `::warning::` and `exit 0` (treat "cannot verify" the same as "verified stale": stop cancelling +further runs, but let the job, and the actual review later in it, proceed). Reproduced the crash against +the pre-fix step with a hand-rolled fake `gh`, confirmed `exit 0` post-fix with the identical fake-failure +fixture, and confirmed the normal (non-failure) cancellation path is unchanged, before folding both +scenarios into `tests/test_noema_review_gate.py` as +`test_superseded_cleanup_survives_a_transient_live_head_lookup_failure`, executing the real, unmodified +production bash (not a reimplementation) via `subprocess.run`, in the same fake-`gh`-fixture idiom +`test_superseded_cleanup_preserves_current_and_newer_run_ids` already established for this step. +`test_noema_concurrency_and_live_head_cleanup_preserve_current_review` was also extended with a docstring +enumerating the four invariants this mechanism now holds together across every review round it took to get +here (new-head cancels old-head; a delayed workflow_run/repository_dispatch trigger never reaches this +step at all; a directional ordering guard stops an older cleanup from racing a newer run; and this +live-head re-check itself fails safe) plus structural assertions for the step's `pull_request_target`-only +gate and the now-guarded (non-bare) live-head re-check -- so a future edit that reintroduces any of these +regressions fails a test immediately rather than requiring another bot-finds-it/human-fixes-it round. + +Validation: `coverage run -m pytest tests -q` -- 2179 passed, 1 skipped, 21 subtests passed (1 new test +plus one extended existing test); `coverage report` -- 100% on `scripts/ci/` (no `.py` production file +touched by this specific fix; the fix and its tests are entirely in `.github/workflows/noema-review.yml`, +`docs/`, and `tests/` -- separately, the unreachable type branch in `extract_json_object` was removed so +the implementation now directly reflects the JSON grammar guarantee); `interrogate` -- 100% docstring +coverage (minimum 100.0%, actual 100.0%); `actionlint` +on the modified workflow -- clean. The touched `run:` block parses with `bash -n` and was exercised +interactively against hand-rolled fake `gh` fixtures for both the crash-reproduction and the fixed +behavior before being folded into the pytest suite. Full validation was re-run after every rebase, given +the branch's ongoing, very high commit velocity from multiple simultaneous sessions converging on this +same ~15-line mechanism throughout the day. + +PR: ContextualWisdomLab/.github#1507 (Devin review on #1507; same PR, addressed before merge). + +The same exact-head review also identified that scanning every opening brace could recover a valid +nested object after its malformed outer object failed to decode. Recovery now considers only top-level +brace groups, preserving lightly wrapped and multiple-object responses while failing closed on nested +escape. A regression test reproduces the former nested-object acceptance directly. An explicit, +string-aware `MAX_JSON_NESTING_DEPTH = 100` check also runs before `raw_decode`, so the limit does not +depend on Python-version-specific `RecursionError` behavior. + +The two chained required-workflow pollers were then replaced after live organization evidence showed +53 concurrent Actions runs and a growing runner queue. The required workflow still dispatches the same +bounded multi-hour OpenCode path and still fails closed without a formal exact-head receipt, but it now +releases its runner after one receipt lookup. Once the privileged dispatch validates the formal receipt, +it selects the latest exact-head `Required OpenCode Review` `pull_request_target` run and calls +`rerun-failed-jobs`; only the small verdict job reruns. This preserves ruleset `18156473`'s required +workflow identity and the two-hour-plus model allowance while removing roughly eleven runner-hours of +polling per PR. The authenticated dispatch carries the immutable triggering required-run ID; the +continuation fetches that target-repository run directly and validates its `pull_request_target` event, +central workflow path, and live PR `head_sha` before rerunning it. This remains correct even when runner +queue delay exceeds the model jobs' declared timeout sum and avoids dependence on context-specific title +or `workflow_url` rendering. Scheduler review retries propagate the same immutable run ID from the +required check's Actions details URL, so the scheduler and direct required-workflow entrypoints share one +continuation contract. Native wake calls use the privileged dispatch job's narrowly scoped `actions: +write` workflow token. Sibling wake calls require `PR_REVIEW_MERGE_TOKEN` or +`OPENCODE_APPROVE_TOKEN` and fail closed when neither is configured; the review-only OpenCode app token +and the central repository's workflow token are never presented as cross-repository Actions credentials. + +## 2026-08-31 `ORCHESTRATOR_PIN_SHA` bumped to carry #925's stream_options/tools fix + +**Context**: `#1451` fixed a separate, org-wide `pingora_edge_policy.py` coverage +gap blocking `opencode-review-dispatch.yml`'s own `coverage-evidence` job for +every `.github`-hosted PR. Once that landed and Strix could actually complete +scans again (via `#1448`'s scoped `LLM_DISABLE_STREAMING` workaround), +`ContextualWisdomLab/contextual-orchestrator#925` — the real root-cause fix for +the gateway's `stream_options.include_usage=true` + `tools` rejection — merged +(`7944a3c`). `.github#1463` reverts `#1448`'s workaround now that the gateway +itself no longer rejects that combination. + +**Devin Review correctly caught a real bug in that revert before merge**: the +review sidecar vendors `contextual-orchestrator` at a *pinned* SHA +(`ORCHESTRATOR_PIN_SHA`), not live `main` — and the pin in place at revert time +(`30c6d71680e659f25a0a433d4726ad0d437f9757`) was cut *before* `#925` merged. +Confirmed by `git merge-base --is-ancestor 30c6d716... 7944a3c` (true). Removing +the Strix-side streaming workaround while the vendored gateway still ran the +old, rejecting code would have restored the exact failure `#1448` existed to +route around — every Strix scan through the sidecar would fail again. + +**Fix**: bumped `ORCHESTRATOR_PIN_SHA` to `7944a3cd98f7b60fba9272e7f89c3977a75af746` +(the `#925` merge commit itself — deliberately not `contextual-orchestrator`'s +later tip, to keep this bump minimal and scoped to exactly the fix this revert +depends on) in the three places this repo's own convention requires kept in +sync: `scripts/ci/contextual_orchestrator_review_sidecar.sh`'s default, +`tests/test_contextual_orchestrator_review_sidecar_contract.py`'s pinned-SHA +contract assertion, and `docs/adr/0003-contextual-orchestrator-vendored-free-zdr.md`'s +"today" reference. Landed in the same PR (`#1463`) as the streaming revert, +not split out, since the revert is unsafe without it. + +## 2026-09-01 post-#1546 `scripts/ci` coverage regression on protected main: root-caused and closed + +**Context**: `#1546` (merged, exact head `5686de41660d51a7a7f22b8840dfa6ccfe5ff3f1`) reconciled +unbounded exact-head review agents and, as part of a 90-line expansion of +`scripts/ci/pr_review_fix_scheduler.py`, added a `live_head_matches` helper, a no-active/no-stale +fall-through branch in `prepare_autofix_slot`, and an "already queued or running" wait branch in +`inspect_pr` — none of which any test exercised directly. This compounded a narrower, older gap in +the same file (`inspect_pr`'s conflicted-draft and conflicted-unauthorized returns) and in +`scripts/ci/pr_review_merge_scheduler.py::fetch_workflow_names_by_check_suite_rest` (pagination, +missing-suite-id/blank-name filtering, non-access-error propagation), first found and attempted in +now-closed, unmerged `#1547`/`#1551`/`#1554` — none of whose evidence or diffs transferred here; +this pass re-derived the current gap from a clean `origin/main` clone rather than assuming those +predecessors were still accurate against `#1546`'s shifted line numbers and new branches. Verified +directly: `coverage report --show-missing` on unmodified `main` showed +`scripts/ci/pr_review_fix_scheduler.py` at 97% (missing 116-121, 459->466, 495, 503, 546) and +`scripts/ci/pr_review_merge_scheduler.py` at 99% (missing 1003, 1008->1005, 1012) — total repo-wide +99%, below the `pyproject.toml` `fail_under = 100` gate. Because `opencode-review-dispatch.yml`'s +`coverage-evidence` job measures the **merged** PR tree (base + head) and hard-fails below 100%, +every PR rebasing onto main inherited this failure regardless of its own diff — org-wide impact, +not scoped to one PR. + +**Fix**: `#1567` (test-only, no production code) adds direct unit coverage for `live_head_matches` +(case-insensitive match, mismatch, malformed-payload paths), `prepare_autofix_slot`'s empty-run +fall-through, the `inspect_pr` conflicted-draft/conflicted-unauthorized/already-queued cases, and +the `fetch_workflow_names_by_check_suite_rest` pagination/filtering/error-propagation paths. +Verified on the fix commit (`db106d50f2134ece147bc5318e389aeb124d198c`): `coverage run -m pytest +tests -q` (2251 passed, 1 skipped, 21 subtests), `coverage report` (repo-wide 100%, both files +individually 100% statement and 100% branch), `interrogate` (100.0%). + +**Devin Review raised a false positive on the fix itself**, claiming +`test_live_head_matches_compares_case_insensitively_and_fails_closed` left non-object-payload, +non-string-SHA, and wrong-length-SHA branches uncovered. Re-verified against the actual gate rather +than accepted at face value: `live_head_matches` has exactly one `if` statement (two arcs, both +exercised by the committed test), and its final `return (isinstance(...) and len(...) == 40 and +...)` is a single boolean expression with no `if`/`else` of its own — `coverage.py`'s branch mode +(what `fail_under = 100` actually measures here) tracks control-flow arcs between statements, not +sub-clause condition coverage within one expression. The cited cases are additional test +thoroughness, not something the gate is currently failing on; confirmed by a full-suite run on the +exact same head showing both files at 100% branch coverage with zero missing branches. Replied with +this evidence on the review thread and did not widen the PR's diff for a claim that does not hold +against this repo's own tooling. + +**One test in the full suite remained a known, pre-existing flake**, unrelated to this change: +`tests/test_opencode_required_verdict_regression.py::test_scheduler_wake_reuses_trusted_receipt_predicate` +intermittently exited 141 (SIGPIPE) under full-suite parallel load; reproduced identically on +unmodified `origin/main` and passed cleanly in file isolation. Not remediated in this pass — out of +scope for a coverage-gap-only PR, and not itself a coverage regression. **Since remediated** (`9e0c0224`, +`fix(test): eliminate scheduler-wake SIGPIPE flake`): the fixture's fake `gh dispatches` responder now +drains its stdin (`cat >/dev/null`) before recording the call, closing the unread-pipe race that +produced the intermittent SIGPIPE (Devin Review, PR #1500). + +## 2026-09-01 naruon#1486 transport-crash: root cause, owner, status + +**Live incident**: the required `noema-review` check on `ContextualWisdomLab/naruon#1486` crashed with an +unhandled `urllib.error.HTTPError: HTTP Error 502: Bad Gateway`. Root cause: `call_llm` in +`scripts/ci/noema_review_gate.py` had `with opener.open(request) as response:` sitting outside the +`try`/`except` that only guarded the JSON-decode/validation steps *after* a successful response -- +identical in shape to, but a distinct bug from, the malformed-verdict crash fixed in `#1507` +(2026-08-31 entries above). Confirmed via direct fetch that `#1546`'s own `call_llm` (main tip at the +time, `5686de41`) carried the same unguarded line, so this crash is orthogonal to, and survives +regardless of, the `#1438`/`#1546` wall-clock-deadline policy question -- `#1438` was closed by the +repo owner as a stale mixed branch unrelated to this specific bug. + +**Fix, round 1**: widened the `try` to cover the request itself and added `urllib.error.URLError` +alongside `RuntimeError` to the existing repair-retry `except` clause -- one retry on a transient +transport failure, then a clean `RuntimeError` on a second failure, matching the malformed-verdict +path's contract. RED (`HTTPError: Bad Gateway` reproduced uncaught) confirmed before, GREEN after. + +**Fix, round 2 (Devin Review, then owner confirmation, on `#1566` itself)**: Devin correctly found that +`response.read()` can raise `http.client.IncompleteRead` -- and, more generally, any +`http.client.HTTPException` or raw `OSError` (a bare socket timeout/disconnect reaching `opener.open()` +before urllib gets a chance to wrap it as `URLError`) -- none of which are `RuntimeError` or +`urllib.error.URLError`, so they still escaped the round-1 boundary. The owner's review comment and +follow-up issue comment on `#1566` confirmed this independently and specified the exact contract: widen +to the bounded transport/read exception families without swallowing JSON/validator/programming errors, +add RED->GREEN regressions for a truncated-body success-after-retry and a repeated-failure case, and at +least one timeout/disconnect family exercising a distinct exception path -- while preserving `#1546`'s +unbounded inference semantics (no fixed inference timeout, no direct-provider fallback, no bypass). + +Widened the `except` clause to `(RuntimeError, urllib.error.URLError, http.client.HTTPException, +OSError)` and simplified the repair-retry re-raise from an `isinstance(exc, urllib.error.URLError)` +check to `isinstance(exc, RuntimeError)`: re-raise as-is only when the second failure is already this +module's own `RuntimeError` (a malformed verdict, an invalid finding, etc.); otherwise wrap in a clean +`RuntimeError`. This generalizes the fail-closed contract to any transport exception type without +needing another `isinstance` branch added per exception class encountered. Three genuinely distinct +exception paths are now each covered by their own RED->GREEN success-after-retry and repeated-failure +regression pair (`test_call_llm_repairs_once_after_a_transport_error_then_succeeds` / +`test_call_llm_fails_closed_after_a_repeated_transport_error` for `HTTPError`/`URLError`; +`test_call_llm_repairs_once_after_a_truncated_response_then_succeeds` / +`test_call_llm_fails_closed_after_a_repeated_truncated_response` for `http.client.IncompleteRead`; +`test_call_llm_repairs_once_after_a_socket_timeout_then_succeeds` / +`test_call_llm_fails_closed_after_a_repeated_socket_timeout` for a raw `TimeoutError` reaching +`opener.open()` directly) -- each verified genuinely RED against the pre-fix boundary before being +folded in, never transferred from an earlier case as substitute proof. Full suite: 2252 passed, 1 +skipped, 21 subtests; `noema_review_gate.py` at 100% line/branch coverage; 100% docstring coverage. + +**Fix, round 3 (Devin Review again, same `#1566`)**: a fourth, distinct bug in the fix itself -- +gating the retry-vs-fail-closed decision on `repair_error`'s truthiness conflated "is this the +second attempt" with "does the caught exception have display text". Several transport exceptions +(a bare `OSError()`/`TimeoutError()`, or an `http.client.HTTPException` raised with no message) all +stringify to `''`, so an empty-message failure on the *first* attempt would leave `repair_error` +falsy on the recursive call too -- the retry-state signal was lost, and `call_llm` would retry +unboundedly (each recursive call itself another live-gateway request) rather than failing closed +after one attempt, eventually crashing on an uncaught `RecursionError` once the interpreter's call +stack was exhausted. Added an explicit `is_retry: bool = False` parameter to track retry state +independently of the exception's text; it (not `repair_error`) now gates both the prompt-injection +branch (falling back to a generic message when `repair_error` is empty) and the except clause's +retry-vs-fail-closed decision, and is threaded through as `is_retry=True` on the recursive call. +Verified genuine RED with a bounded-recursion regression test +(`test_call_llm_fails_closed_after_a_repeated_empty_message_transport_error`, which raises a +diagnostic `AssertionError` if `call_llm` retries more than once instead of letting it recurse to +CPython's own limit) before this fourth fix, GREEN after -- paired with +`test_call_llm_repairs_once_after_an_empty_message_transport_error_then_succeeds` for the +happy-path case. Full suite: 2254 passed, 1 skipped, 21 subtests; `noema_review_gate.py` still at +100% line/branch coverage, 100% docstring coverage. + +**Owner**: this repo (`ContextualWisdomLab/.github`), `scripts/ci/noema_review_gate.py`. +**Status**: fixed on `ContextualWisdomLab/.github#1566` (branch `fix/noema-review-transport-error-retry`), +pending required checks and final review. + +While verifying this fix's full-suite run, an unrelated, pre-existing SIGPIPE (exit 141) flake was also +found and root-caused in `tests/test_opencode_required_verdict_regression.py::test_scheduler_wake_reuses_trusted_receipt_predicate`: +its fake `gh` fixture never drains the JSON piped into it via `--input -` for the dispatch call, so under +`set -euo pipefail` the pipeline's writer (`jq`) can be killed by `SIGPIPE` if the fake reader exits +first -- reproduced locally at roughly a 60% failure rate over 15 runs in complete isolation (not merely +under CI load), and eliminated (30/30 clean runs) by draining stdin (`cat >/dev/null`) before the fixture +writes its own output. Fixed separately, since it is unrelated to the transport-crash file above; see +that PR for its own evidence. + +## 5. 실행 루프와 고객의 다음 행동 + +각 hourly pass는 아래 순서를 유지한다. + +1. 조직·repo 책임 경계를 확인하고, current default branch SHA와 PR head SHA를 새로 읽는다. +2. 열린 PR 하나를 선택해 review threads, formal review commit SHA, required Checks와 failure logs를 확인한다. +3. 실패가 코드 결함이면 root cause를 해당 PR의 최소 범위에서 수정하고, 원격 agent의 concurrent commit은 normal forward history로 보존한다. Force-push하지 않는다. +4. 현실적인 domain test, edge test, docstring/branch coverage, security/SBOM, actionlint/browser evidence를 실행한다. +5. 새 head에서 Checks를 재실행하고 independent current-head approval을 다시 요청한다. OpenCode/Strix/Noema 지연은 blocker가 아니다. 기다리는 동안 다음 PR 또는 Gap을 진행한다. +6. protected ruleset의 approval·resolved thread·terminal Checks·exact head를 모두 충족할 때만 `--match-head-commit` normal merge한다. 조건이 안 되면 merge하지 않고 다음 PR로 진행한다. +7. PR이 소진되면 Project #1과 소비 repo에서 가장 큰 운영자/제품 Gap을 선택해 새 PR을 만들고, 이 문서의 Gap ID를 연결한다. 다음 제품 increment의 소유 저장소는 naruon(G-06/G-15)이다. + +운영자는 receipt의 `next_action`만 실행하면 된다. `PR_REVIEW_MERGE_TOKEN` 부재나 provider/runner 지연은 token 값을 로그에 남기지 않고 원인을 기록한 뒤 다음 hourly pass에서 exact head를 재검증한다. + +`COPILOT_GITHUB_TOKEN`은 사용하지 않는다. 기존 리뷰용 Agent 키 체계는 유지한다. + +### 5.1 이번 루프의 다음 개발 increment + +1. ContextualWisdomLab/.github#1297 — current-head Strix serialization과 scoped close cleanup의 hosted Checks·독립 승인을 재확인한 뒤 보호된 auto-merge를 기다린다. +2. ContextualWisdomLab/.github#1345/#1347 — 각각 normalizer 선형 스캔과 web-E2E isolation/SSRF 수정의 terminal Checks·Strix·Noema 증거를 같은 HEAD에서 재확인한다. +3. ContextualWisdomLab/.github#1326 — Appguardrail/macOS hourly caller를 current CodeRabbit finding 및 APA citation evidence와 함께 재검토한다. +4. G-01/G-02는 중앙 control-plane merge evidence의 current-head 품질 문제, G-05/G-06는 naruon ecosystem 소비 증거, G-15는 대용량·미지원 첨부파일 parser registry의 소유 저장소 PR로 연결한다. +5. `scripts/ci/select_nvidia_nim_model.py`(호출자 없음, 위 §5의 여러 항목이 이미 문서화)를 별도의 작은 PR(`fix/remove-orphaned-nim-model-resolver`)로 분리 제거했다 — `#1437` 리뷰 스레드가 명시적으로 요청한 대로 direct-NIM cleanup을 pool-flip 논의와 분리했다. `contextual_orchestrator_review_sidecar.sh`의 참조 주석은 git history를 가리키도록 갱신했다. + +## 6. Compliance and data boundary + +- PII 원문을 무조건 masking하여 업무를 끊지 않는다. 대신 purpose-bound access lease, field-level encryption/tokenization, consented minimal-disclosure consequence, audited access, revocation/deletion을 사용한다. `COPILOT_GITHUB_TOKEN`은 사용하지 않는다. +- 모델·리뷰·sandbox·Checks·merge·release는 서로 다른 authority다. 하나의 PASS를 approval이나 release로 승격하지 않는다. +- 모든 untrusted input, repository patch, image/base64 payload, model output은 data로 취급하고 command/credential로 해석하지 않는다. +- demo/synthetic fixture는 unit test에만 두며 production seed/fixture에는 포함하지 않는다. +- CSAP and SOC 2 evidence maps belong with consent/lease/tokenization, not blanket PII masking. + +## 7. APA 7th references + +American Institute of Certified Public Accountants. (2017). *2017 trust services criteria for security, availability, processing integrity, confidentiality, and privacy*. AICPA. + +International Organization for Standardization. (2022). *ISO/IEC 27001:2022 information security, cybersecurity and privacy protection—Information security management systems—Requirements*. ISO. + +International Organization for Standardization. (2023). *ISO/IEC 42001:2023 information technology—Artificial intelligence—Management system*. ISO. + +National Institute of Standards and Technology. (2023). *Artificial intelligence risk management framework (AI RMF 1.0)* (NIST AI 100-1). U.S. Department of Commerce. https://doi.org/10.6028/NIST.AI.100-1 + +World Wide Web Consortium. (2023). *Web Content Accessibility Guidelines (WCAG) 2.2*. https://www.w3.org/TR/WCAG22/ + +Lewis, P., Perez, E., Piktus, A., Petroni, F., Karpukhin, V., Goyal, N., Küttler, H., Lewis, M., Yih, W.-t., Rocktäschel, T., Riedel, S., & Kiela, D. (2020). Retrieval-augmented generation for knowledge-intensive NLP tasks. *Advances in Neural Information Processing Systems, 33*, 9459–9474. + +Tang, Y., Cetin, E., Xu, J., Sun, Q., Nielsen, S., Richard, V., Goda, H., Tymchenko, I., Nguyen, N., Lee, H., Ashiga, M., Kotyan, S., Kuroki, S., & Clanuwat, T. (2026). *Sakana Fugu technical report* [Technical report]. arXiv. https://doi.org/10.48550/arXiv.2606.21228 + +Zhang, S., Yu, Y., Li, Y., Zhao, W., Yang, Y., Zhang, Y., & Liu, T. (2025). *Conductor: Learning to route multi-agent workflows* [Preprint]. arXiv. https://doi.org/10.48550/arXiv.2512.04388 + +Xu, J., Sun, Q., Schwendeman, P., Nielsen, S., Cetin, E., & Tang, Y. (2026). *TRINITY: An evolved LLM coordinator* [Preprint]. arXiv. https://doi.org/10.48550/arXiv.2512.04695 + +Higgins, S. S., Crepalde, N., & Fernandes, L. (2021). Segmented multiplexity: A research agenda for multiplexity beyond the average. *PLOS ONE, 16*(9), e0257527. https://doi.org/10.1371/journal.pone.0257527 + + +## Noema reviewer credential-lifetime delta — 2026-09-01 + +**Observed gap.** `ContextualWisdomLab/naruon#1497@152d1998c4e8024be9dc7026c8789d343c884fd0` demonstrated a control-plane latency/authority defect: a repository-scoped `cwl-noema-review` GitHub App token minted before contextual-orchestrator model work expired before the next GitHub operation, producing HTTP 401 even though repository-owned deterministic checks were otherwise successful. This is a central `.github` reviewer-lifecycle gap, not a Naruon product failure. + +**Owner-side closure in #1616.** The Noema workflow now treats model preparation and GitHub publication as separate trust phases. A bounded private envelope carries only the model verdict; the GitHub App path remints the same repository-scoped least-privilege authority after model work, and publication independently verifies repository, PR number, canonical exact head, live PR state, draft state, independent reviewer actor, and duplicate-current-head review state before submission. No predecessor-head evidence or predecessor App credential is accepted as publication authority. PAT/OIDC remain explicit sources and there is no `github.token` or author fallback. + +**Executable evidence.** `tests/test_noema_reviewer_token_lifetime.py` binds the production workflow step graph to prepare → fresh App mint → publish with exact-head arguments and source-specific credentials. `tests/test_noema_two_phase_handoff.py` executes the helper against controlled gate doubles and proves no preparation-side publication, fresh-head/actor rebinding, stale-head non-publication, draft skip behavior, cleanup on malformed handoff, and hard-link alias rejection. `.github/workflows/noema-token-lifetime-quality-ci.yml` runs these contracts with hash-pinned dependencies on every relevant seam. + + +**Regression-suite consistency.** Legacy broader-suite assertions that still named the retired single-process Noema step/module are migrated to the two-phase prepare/publish contract, including step-scoped helper and envelope-argument evidence. This closes the false-GREEN gap where focused token-lifetime CI could pass while unchanged broader contracts described an impossible execution path. + +**Residual external verification.** After this central change reaches protected `main`, replay Required Noema Review for unchanged `naruon#1497@152d1998c4e8024be9dc7026c8789d343c884fd0`. Closure evidence requires a current-head schema-valid review or typed review-unavailable outcome without expired-token 401; a pre-merge run cannot prove the merged workflow-source path and is not promoted to release evidence. + + +## 2026-09-01 central required review workflows: floating runner image contributing to organization-wide queuing + +**Observed gap.** `#1618` (required security gates) and `#1609` (merge scheduler) already pinned their jobs off `ubuntu-latest` after this session found it to be, in that fix's own words, "the observed starved floating image" — GitHub-hosted runners requesting the floating `ubuntu-latest` label were being left `queued` with no runner assignment for hours, well beyond ordinary scheduling latency, while identical jobs on other repositories/workflows completed normally. `strix.yml`, `opencode-review.yml`, and `noema-review.yml` — the three workflows the org's own required-workflow ruleset runs against every PR in every sibling repository — still requested `ubuntu-latest` on every job (9 occurrences total: 3 in `strix.yml`, 5 in `opencode-review.yml`, 2 in `noema-review.yml`; `pr-review-merge-scheduler.yml` was already covered by `#1609`). Since these three are the actual required-check gate blocking merge across the whole organization, a starved image here is a direct, high-leverage contributor to the sustained multi-hour organization-wide queuing observed throughout this session (independently corroborated by `#1630`'s own record of 822 queued Actions runs at merge time). + +**Fix.** Pinned all 9 occurrences to the explicit `ubuntu-24.04` image, matching the pattern already established by `#1618`/`#1609` exactly (a literal `runs-on:` value swap, no other job semantics touched). New `tests/test_required_review_runner_image_contract.py` asserts no job in any of the three files requests the floating image and pins the expected per-file occurrence count, mirroring `test_required_security_runner_image_contract.py`'s existing structure. + +**Unrelated pre-existing failures fixed in the same pass.** `#1630` (merged shortly before this fix, itself an owner-authorized `QUEUE_SATURATION_CHICKEN_EGG` bypass addressing the same 822-run backlog) moved the organization sweep's rotation cadence from every 15 minutes to hourly to reduce control-plane pressure, changing `pr-review-merge-scheduler.yml`'s `ORG_SWEEP_ROTATION_INDEX` wall-clock fallback divisor from `900` (15 minutes in seconds) to `3600` (1 hour), but left `tests/test_required_workflow_queue_contract.py`'s four rotation-index tests asserting the old `900` divisor and the old literal workflow string. Confirmed these 4 failures reproduce identically on a clean `origin/main` checkout with no changes from this branch, independent of and pre-dating this fix. Updated all four to the new `3600` divisor/string, preserving each test's original intent (wall-clock fallback on total counter unavailability, transient-read-failure-does-not-reset, successful-read-but-failed-patch-falls-back, and the documentation/input-validation contract) unchanged. + +**Validation.** Full suite `2407 passed, 1 skipped, 21 subtests`; `coverage` 100% on `scripts/ci`; `interrogate` 100%; all four touched/added workflow files re-parse as valid YAML; `test_opencode_workflow_shell_syntax.py` and related shell-syntax tests pass unchanged. + +**Residual.** This closes the specific floating-image contribution from these three central workflows; it does not by itself guarantee the organization-wide Actions queue is fully drained, since other repositories' own workflows and any remaining unpinned central workflows may still request the floating image. Worth a follow-up sweep across the rest of `.github/workflows/` and sibling-repo workflows if queuing persists after this lands. + +## 2026-09-02 GitHub Actions review sidecar pool pinned to `orchestrator/free`; `auto` removed as an accepted value + +**Problem.** `scripts/ci/contextual_orchestrator_review_sidecar.sh` — the script every central required review workflow (Strix, OpenCode Review, Noema Review, the PR-review autofix sidecar) provisions to talk to `contextual-orchestrator` — read an operator-settable `CONTEXTUAL_ORCHESTRATOR_POOL` environment variable, defaulted it to `free`, and validated it against exactly two accepted values: `free` or `auto` (`case "$orchestrator_pool" in free|auto) ...`). `auto` is a real, load-bearing value one layer down: `scripts/ci/contextual_orchestrator_review_launcher.py --pool auto` admits *priced* discovered routes as a fallback stage once the free pool is exhausted (`build_zdr_prioritized_catalog(..., pool="auto")`), by design, for callers that want that behavior. Nothing in this repository's own review-provisioning code path currently sets `CONTEXTUAL_ORCHESTRATOR_POOL=auto` — the only workflow that sets the variable at all, `strix.yml`, sets it to `free`; every other central review workflow simply relies on the script's own `:-free` default — so this was not a live incident, it was an unaudited, structurally-reachable escape hatch: a future edit to any of the four workflows above, or a manually-triggered `workflow_dispatch` with a custom env override, could set `CONTEXTUAL_ORCHESTRATOR_POOL=auto` and the sidecar would accept it silently, with no cost ceiling, no budget/authorization gate, and no reviewer visibility that priced models were now in scope for a required check. + +**Why this matters now, not hypothetically.** The org's explicit standing operating directive (the perpetual PR review→fix→merge→develop loop this session runs under) states plainly that the free+ZDR routing combination is not yet solved reliably in central CI — this exact gap-baseline document's own accumulated 2026-08-30/08-31 entries above record a real `orchestrator/free` exhaustion incident, a crowding-out bug between shared-endpoint credentials, and multiple rounds of Devin-Review-caught admission-priority defects in `contextual_orchestrator_review_policy.py`, all specifically about getting the *free* pool right. Admitting a priced-inclusive `auto` pool into required review workflows before that work is solid would let one misconfiguration or one well-intentioned "let's widen coverage" workflow edit start spending real provider credit on every PR's required Strix/OpenCode/Noema review, with no operator-visible signal that this had happened — the sidecar's own `log` lines print the resolved pool, but nothing downstream alerts on it, and there is no spend cap in this repository's own review-provisioning path (unlike `contextual-orchestrator`'s own cost-ledger, which this vendored sidecar path does not call into for CI review spend). + +**Alternatives considered.** +1. *Leave `auto` accepted but never set it.* Rejected: this is the status quo, and the status quo is exactly the unaudited escape hatch described above — "nobody currently sets it" is not a control, it is an absence of one. +2. *Remove the `CONTEXTUAL_ORCHESTRATOR_POOL` environment variable entirely, hard-coding `--pool free` with no override mechanism.* Considered and rejected in favor of the fail-closed `case` statement kept below: removing the variable removes the ability to reason about *why* an override was rejected (a caller setting `auto` would instead see an unrelated "unrecognized flag" or `--pool` argparse error further downstream, or silently fall through to whatever the launcher's own default resolves to, depending on how the removal was implemented) and removes a natural place to extend validation later (e.g. if the org ever explicitly re-authorizes `auto` for CI with a budget gate, only this one `case` arm needs to change). A `case` statement that explicitly names and rejects `auto` with a clear diagnostic is this repository's own established idiom (see the sibling `CONTEXTUAL_ORCHESTRATOR_REQUIRE_ZDR` validation two lines above it in the same file) and is more auditable, not less. 3. *Narrow the launcher's own `--pool` argparse choices to just `("free",)`.* Rejected: the launcher (`contextual_orchestrator_review_launcher.py`) is a general-purpose CLI, not GitHub-Actions-specific — it is invoked directly (outside any workflow) for local testing and by other, non-CI-review callers that may have a legitimate reason to exercise the `auto` pool's priced-fallback behavior. Narrowing it there would remove functionality the tool's own design intentionally provides, contradicting the directive's explicit scoping ("GitHub Actions Workflow 이용에 관해" — regarding GitHub Actions Workflow *usage* specifically, not the tool in general). `test_launcher_uses_orchestrator_discovery_and_governed_pools`'s existing pin of `choices=("free", "auto")` on the launcher was therefore left unchanged. **Fix.** `scripts/ci/contextual_orchestrator_review_sidecar.sh`'s `case "$orchestrator_pool" in` now accepts only `free`; every other value (`auto` included, and any typo/unexpected value) falls to the `*)` arm and calls `fail "CONTEXTUAL_ORCHESTRATOR_POOL must be free"`, matching this script's own existing fail-closed idiom for `CONTEXTUAL_ORCHESTRATOR_REQUIRE_ZDR`. The variable's default (`${CONTEXTUAL_ORCHESTRATOR_POOL:-free}`) is unchanged, so every existing caller (all of which already resolve to `free`, explicitly or by default) is unaffected — this is a pure narrowing of previously-unused surface, not a behavior change for any current workflow run. From 7078e9293c5c498ba74ac2f60ad91e92e937665c Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 05:02:56 +0900 Subject: [PATCH 34/43] docs(gap): record baseline truncation RCA --- docs/doctoring/opencode-same-model-midabort-20260919.md | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/docs/doctoring/opencode-same-model-midabort-20260919.md b/docs/doctoring/opencode-same-model-midabort-20260919.md index 181942116e..44f2468444 100644 --- a/docs/doctoring/opencode-same-model-midabort-20260919.md +++ b/docs/doctoring/opencode-same-model-midabort-20260919.md @@ -1,5 +1,14 @@ # OpenCode same-model mid-abort and context loss (2026-09-19) +## Cross-file Gap baseline preservation repair + +Documentation successor `33e86ea6becb754e6e3b8a98299640220ecf27d5` accidentally treated a truncated GitHub contents response as the complete `docs/product-technical-gap-baseline.md`: the diff was `+7/-1,505`, the literal truncation diagnostic entered the file, and protected level-two authority fell from 59 sections to 36. That was not an intentional supersession and was unrelated to the checkpoint runtime repair. + +RED `117c1bf07ab07b8a7262da3d2da9d1f437c36cf0` integrates the canonical-owner preservation contract already established on #2281 and fails on five missing representative security, runtime, compliance, APA, and credential-lifetime sections. GREEN `e55c34159b43bd32fa7e039d69faa2b5f0ab5814` reconstructs the exact `7c5844ad…` parent baseline from bounded line ranges, then applies only the three intended OpenCode control-row updates. The repaired file has 59/59 protected level-two sections, no truncation marker, one instance of every OpenCode control row, and a `+3/-3` baseline diff relative to `7c5844ad…`. + +This test is a cross-file authority guard, not checkpoint behavior evidence. It prevents a future documentation-only successor from erasing unrelated PRD/TRD, Context Map, security, runtime, compliance, or APA decisions while preserving the checkpoint lane's valid delta. + + ## Scope Improve OpenCode stopping mid-work or misunderstanding required outputs on From 4642b09dda7919a5c23a0a38e658335529d2716a Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 05:58:43 +0900 Subject: [PATCH 35/43] test: reject assistant-authored checkpoint causes --- ...test_opencode_review_session_checkpoint.py | 28 +++++++++++++++++++ 1 file changed, 28 insertions(+) diff --git a/tests/test_opencode_review_session_checkpoint.py b/tests/test_opencode_review_session_checkpoint.py index 9975163c5e..dea5bbd6a2 100644 --- a/tests/test_opencode_review_session_checkpoint.py +++ b/tests/test_opencode_review_session_checkpoint.py @@ -78,6 +78,34 @@ def test_classify_termination_detects_provider_fatal(tmp_path: Path) -> None: assert reason == "provider-fatal" +def test_classify_termination_ignores_assistant_prose_for_cause( + tmp_path: Path, +) -> None: + """Assistant prose cannot author a trusted provider termination label.""" + json_path = tmp_path / "run.jsonl" + json_path.write_text( + '{"type":"step_start","sessionID":"session-1"}\n', + encoding="utf-8", + ) + export_path = tmp_path / "export.json" + export_path.write_text( + _export_with_text( + "Review finding mentions ContextOverflowError, timeout, rate limit, " + "model_not_found, and invalid-control-output." + ), + encoding="utf-8", + ) + + reason = classify_termination( + json_path=json_path, + export_path=export_path, + exit_code=3, + log_hint="invalid-control-output", + ) + + assert reason == "invalid-control" + + def test_record_and_continue_excludes_provider_identity_from_prompt(tmp_path: Path) -> None: """Leaf continuation prompts cannot consume provider-specific route telemetry.""" export_path = tmp_path / "export.json" From 3f05c1fc2348ba7c957976dfe31955e00679eb51 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 05:58:56 +0900 Subject: [PATCH 36/43] test: preserve invalid-control checkpoint cause --- tests/test_opencode_model_pool_runner.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_opencode_model_pool_runner.py b/tests/test_opencode_model_pool_runner.py index 2ac8a6bd75..2d50f7800e 100644 --- a/tests/test_opencode_model_pool_runner.py +++ b/tests/test_opencode_model_pool_runner.py @@ -1123,3 +1123,4 @@ def test_same_model_retry_appends_host_checkpoint_continuation(tmp_path: Path) - assert "Same-model continuation (host checkpoint" in prompt_text assert "partial review without control block" not in prompt_text assert "contextual-orchestrator/orchestrator/free" in prompt_text + assert "Termination reason: `invalid-control`" in prompt_text From f57319d18d8e27547d8077fbc2310265753962b7 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 05:59:29 +0900 Subject: [PATCH 37/43] fix: trust host checkpoint causes only --- scripts/ci/opencode_review_session_checkpoint.py | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/scripts/ci/opencode_review_session_checkpoint.py b/scripts/ci/opencode_review_session_checkpoint.py index ba26a4cd4f..de2253d248 100644 --- a/scripts/ci/opencode_review_session_checkpoint.py +++ b/scripts/ci/opencode_review_session_checkpoint.py @@ -65,11 +65,18 @@ def classify_termination( log_hint: str = "", ) -> str: """Return a bounded termination reason for one OpenCode attempt.""" + structured_error_events: list[str] = [] + for line in _read_bounded_text(json_path, 65536).splitlines(): + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + if isinstance(event, Mapping) and event.get("type") == "error": + structured_error_events.append(line) combined = "\n".join( part for part in ( - _read_bounded_text(json_path, 65536), - _read_bounded_text(export_path, 65536), + "\n".join(structured_error_events), _read_bounded_text(stderr_path, 65536) if stderr_path else "", log_hint, ) From 9dc05e7db9362e74ef63e83b9ee3ed1df9ddd116 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 05:59:47 +0900 Subject: [PATCH 38/43] fix: preserve invalid-control checkpoint cause --- scripts/ci/run_opencode_review_model_pool.sh | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/scripts/ci/run_opencode_review_model_pool.sh b/scripts/ci/run_opencode_review_model_pool.sh index 55f8bac3f7..0737bc68b3 100644 --- a/scripts/ci/run_opencode_review_model_pool.sh +++ b/scripts/ci/run_opencode_review_model_pool.sh @@ -198,6 +198,10 @@ record_session_checkpoint() { local export_file="$5" local exit_code="$6" local stderr_file="$7" + local log_hint="" + if [ "$exit_code" -eq 3 ]; then + log_hint="invalid-control-output" + fi PYTHONPATH="$GITHUB_WORKSPACE${PYTHONPATH:+:$PYTHONPATH}" python3 "$GITHUB_WORKSPACE/scripts/ci/opencode_review_session_checkpoint.py" record \ --checkpoint "$checkpoint_file" \ --model-candidate "$model_candidate" \ @@ -208,6 +212,7 @@ record_session_checkpoint() { --json-path "$json_file" \ --export-path "$export_file" \ --exit-code "$exit_code" \ + --log-hint "$log_hint" \ ${stderr_file:+--stderr-path "$stderr_file"} \ || true } From 01ef4c5bbeb381677cb9301e771856c9d9dc9cac Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 06:00:46 +0900 Subject: [PATCH 39/43] docs: record checkpoint cause authority gap --- docs/product-technical-gap-baseline.md | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/product-technical-gap-baseline.md b/docs/product-technical-gap-baseline.md index 1cba9af6aa..935ff724aa 100644 --- a/docs/product-technical-gap-baseline.md +++ b/docs/product-technical-gap-baseline.md @@ -14,6 +14,7 @@ | CONTROL-OPENCODE-CHECKPOINT-INTEGRITY-01 | **Source repaired on PR #2284; hosted exact-head acceptance pending** | Review of `#2284@9fdfddfa` found complete-file reads before slicing, user-prompt marker laundering, accumulated retry appendices, and checkpoint application outside `contextual-orchestrator/orchestrator/free`. Test-only `afe1420d` produced exactly 4 failures; source `b400ad5d` produced 43 focused warnings-as-errors passes. Later test expansion at `9012eac2` hid an arithmetically unreachable `used < 0` decision from coverage while its named test exercised only `used == 0`. RED `802a4fa5` makes that vacuous oracle executable; GREEN `0aa9b902` removes only the impossible clamp. Exact `7c5844ad` (tree `53908068`) passes checkpoint/runner **65 tests** and the full warnings-as-errors suite **3,428 passed / 5 skipped / 40 subtests passed**. | ContextualWisdomLab/.github owns the trusted OpenCode host checkpoint boundary. Keep #2284 Draft until fresh terminal hosted security/quality evidence and qualifying independent review exist; no predecessor result transfers. | | CONTROL-OPENCODE-CONTINUATION-AUTHORITY-02 | **Missing or negative authority now fails closed on PR #2284; calibrated admission remains Proposed** | Exact `bed37694` silently selected budget `2` although controlled completion/time/token evidence was still pending. RED `f0775fd4` and `c4165352` require the runner and direct CLI to reject absent authority; GREEN `3e290447` and `4070c161` remove the Python and shell defaults. Exact-head RED `68459817` then proves both direct CLI paths accepted negative authority, exited zero, and emitted `0`; GREEN `7c5844ad` validates non-negative authority at the parser boundary. Focused **65 passed**, full suite **3,428 passed / 5 skipped / 40 subtests passed**, compileall, Bash syntax, and diff check bind the repair to tree `53908068`. | ContextualWisdomLab/.github owns host enforcement. A budget may be enabled only after a versioned fast-mlsirm/Fugu/Conductor/TRINITY-compatible allocator receipt and controlled A/B evidence are integrated; absent or invalid authority keeps checkpoint injection disabled. Hosted exact-head GREEN and independent review remain required. | | CONTROL-OPENCODE-PROVIDER-NEUTRAL-03 | **Consumer schema copy removed on PR #2284; released CO projection pending** | Exact `bed37694` parsed CO model/provider/phase/status fields and formatted them into continuation prompts. RED `7c5c6a75`/`3a8c056b` requires byte-neutral handling of hostile provider details. GREEN `866cc6a4`/`8c04d128` removes route parsing and runner plumbing; `51a53188`/`b7e8256f` deletes the mutable parser and fixtures. Exact `7c5844ad` retains that provider-neutral boundary and passes checkpoint/runner **65 tests** plus the full warnings-as-errors suite **3,428 passed / 5 skipped / 40 subtests passed**. | ContextualWisdomLab/contextual-orchestrator issue #1106 owns the released provider-neutral allocation receipt. ContextualWisdomLab/.github must know only `orchestrator/free` and the gateway token; provider identities remain CO observability data. Keep Draft until immutable owner release/pin, exact-head GREEN, and independent review. | +| CONTROL-OPENCODE-CHECKPOINT-CAUSE-04 | **Source repaired on PR #2284; hosted exact-head acceptance pending** | Exact `3f05c1fc` proves two causal-context failures: provider-like words in assistant prose selected a trusted provider termination label, and an invalid control result reached the continuation as generic `nonzero-exit` although the host already held wrapper status `3`. GREEN `f57319d1` restricts classification to structured OpenCode `type=error` events, CLI stderr, and fixed host hints; GREEN `9dc05e7d` maps wrapper status `3` to `invalid-control-output`. Exact-source assertions confirm assistant export is no longer causal while export presence remains checked, and the runner continuation contract requires `Termination reason: invalid-control`. | ContextualWisdomLab/.github owns checkpoint cause identity. Keep #2284 Draft until exact-head Python/runner/security gates are terminal GREEN and an independent review qualifies; no assistant prose, prior head, or predecessor check may author the cause. | ### 2026-09-13 current-head incident delta From c8680c560703f4524be430bae92fb8efd8bd3c53 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 06:52:27 +0900 Subject: [PATCH 40/43] test(opencode): bound checkpoint session exports --- ...test_opencode_review_session_checkpoint.py | 32 +++++++++++++++++++ 1 file changed, 32 insertions(+) diff --git a/tests/test_opencode_review_session_checkpoint.py b/tests/test_opencode_review_session_checkpoint.py index dea5bbd6a2..e361f4cdb6 100644 --- a/tests/test_opencode_review_session_checkpoint.py +++ b/tests/test_opencode_review_session_checkpoint.py @@ -46,6 +46,38 @@ def test_summarize_partial_assistant_never_returns_raw_text(tmp_path: Path) -> N assert "secret" not in json.dumps(summary) +def test_oversized_session_export_fails_closed_without_full_parse( + tmp_path: Path, +) -> None: + """Provider session exports above the owner evidence bound are not parsed.""" + export_path = tmp_path / "oversized-export.json" + oversized_text = "x" * (2 * 1024 * 1024 + 1) + "opencode-review-control-v1" + export_path.write_text(_export_with_text(oversized_text), encoding="utf-8") + + summary = summarize_partial_assistant(export_path) + entry = record_attempt_checkpoint( + checkpoint_path=tmp_path / "checkpoint.json", + model_candidate="contextual-orchestrator/orchestrator/free", + attempt=1, + head_sha="a" * 40, + run_id="1", + run_attempt="1", + json_path=tmp_path / "missing.jsonl", + export_path=export_path, + exit_code=1, + ) + + assert summary["assistant_text_present"] is False + assert entry["partial_summary"]["assistant_text_present"] is False + assert entry["missing_required_outputs"] == [ + "opencode-review-control-v1", + "adversarial_validation", + '"result"', + "Developer experience:", + "User experience:", + ] + + def test_missing_required_outputs_lists_absent_markers() -> None: """Incomplete control output records which contract markers are still missing.""" missing = missing_required_outputs("partial progress only") From 5c49d4f80ac94e07868fbe0f3d15a70aceb8b024 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 06:53:41 +0900 Subject: [PATCH 41/43] fix(opencode): bound checkpoint session export reads --- .../ci/opencode_review_session_checkpoint.py | 32 +++++++++++++++---- 1 file changed, 26 insertions(+), 6 deletions(-) diff --git a/scripts/ci/opencode_review_session_checkpoint.py b/scripts/ci/opencode_review_session_checkpoint.py index de2253d248..760e421346 100644 --- a/scripts/ci/opencode_review_session_checkpoint.py +++ b/scripts/ci/opencode_review_session_checkpoint.py @@ -42,6 +42,7 @@ ) MAX_PARTIAL_DIGEST_CHARS = 64 MAX_CONTINUATION_BYTES = 8192 +MAX_SESSION_EXPORT_BYTES = 2 * 1024 * 1024 def _read_bounded_text(path: Path, max_bytes: int) -> str: @@ -56,6 +57,23 @@ def _read_bounded_text(path: Path, max_bytes: int) -> str: return data.decode("utf-8", errors="replace") +def _read_bounded_session_export(path: Path) -> str: + """Read a complete session export within the owner evidence byte bound.""" + if not path.is_file(): + return "" + try: + with path.open("rb") as bounded_stream: + data = bounded_stream.read(MAX_SESSION_EXPORT_BYTES + 1) + except OSError: + return "" + if len(data) > MAX_SESSION_EXPORT_BYTES: + return "" + try: + return data.decode("utf-8") + except UnicodeDecodeError: + return "" + + def classify_termination( *, json_path: Path, @@ -94,7 +112,8 @@ def classify_termination( def summarize_partial_assistant(export_path: Path) -> dict[str, str | int | bool]: """Return bounded metadata about partial assistant output, never raw text.""" - if not export_path.is_file(): + encoded_export = _read_bounded_session_export(export_path) + if not encoded_export: return { "assistant_text_present": False, "assistant_line_count": 0, @@ -102,8 +121,8 @@ def summarize_partial_assistant(export_path: Path) -> dict[str, str | int | bool "has_control_sentinel": False, } try: - payload = json.loads(export_path.read_text(encoding="utf-8")) - except (OSError, json.JSONDecodeError, UnicodeDecodeError): + payload = json.loads(encoded_export) + except json.JSONDecodeError: return { "assistant_text_present": False, "assistant_line_count": 0, @@ -181,9 +200,10 @@ def record_attempt_checkpoint( """Append one bounded attempt record to the host checkpoint ledger.""" partial_summary = summarize_partial_assistant(export_path) partial_text = "" - if export_path.is_file(): + encoded_export = _read_bounded_session_export(export_path) + if encoded_export: try: - payload = json.loads(export_path.read_text(encoding="utf-8")) + payload = json.loads(encoded_export) if isinstance(payload, dict): messages = payload.get("messages") if isinstance(messages, list): @@ -205,7 +225,7 @@ def record_attempt_checkpoint( ): chunks.append(part["text"]) partial_text = "\n".join(chunks) - except (OSError, json.JSONDecodeError, UnicodeDecodeError): + except json.JSONDecodeError: partial_text = "" entry = { "attempt": attempt, From 53b82180ec6c671d5a1ed4c963efdd7d48230b3e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 06:54:46 +0900 Subject: [PATCH 42/43] docs(gap): bind checkpoint export memory boundary --- docs/product-technical-gap-baseline.md | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/product-technical-gap-baseline.md b/docs/product-technical-gap-baseline.md index 935ff724aa..72aabf380b 100644 --- a/docs/product-technical-gap-baseline.md +++ b/docs/product-technical-gap-baseline.md @@ -15,6 +15,7 @@ | CONTROL-OPENCODE-CONTINUATION-AUTHORITY-02 | **Missing or negative authority now fails closed on PR #2284; calibrated admission remains Proposed** | Exact `bed37694` silently selected budget `2` although controlled completion/time/token evidence was still pending. RED `f0775fd4` and `c4165352` require the runner and direct CLI to reject absent authority; GREEN `3e290447` and `4070c161` remove the Python and shell defaults. Exact-head RED `68459817` then proves both direct CLI paths accepted negative authority, exited zero, and emitted `0`; GREEN `7c5844ad` validates non-negative authority at the parser boundary. Focused **65 passed**, full suite **3,428 passed / 5 skipped / 40 subtests passed**, compileall, Bash syntax, and diff check bind the repair to tree `53908068`. | ContextualWisdomLab/.github owns host enforcement. A budget may be enabled only after a versioned fast-mlsirm/Fugu/Conductor/TRINITY-compatible allocator receipt and controlled A/B evidence are integrated; absent or invalid authority keeps checkpoint injection disabled. Hosted exact-head GREEN and independent review remain required. | | CONTROL-OPENCODE-PROVIDER-NEUTRAL-03 | **Consumer schema copy removed on PR #2284; released CO projection pending** | Exact `bed37694` parsed CO model/provider/phase/status fields and formatted them into continuation prompts. RED `7c5c6a75`/`3a8c056b` requires byte-neutral handling of hostile provider details. GREEN `866cc6a4`/`8c04d128` removes route parsing and runner plumbing; `51a53188`/`b7e8256f` deletes the mutable parser and fixtures. Exact `7c5844ad` retains that provider-neutral boundary and passes checkpoint/runner **65 tests** plus the full warnings-as-errors suite **3,428 passed / 5 skipped / 40 subtests passed**. | ContextualWisdomLab/contextual-orchestrator issue #1106 owns the released provider-neutral allocation receipt. ContextualWisdomLab/.github must know only `orchestrator/free` and the gateway token; provider identities remain CO observability data. Keep Draft until immutable owner release/pin, exact-head GREEN, and independent review. | | CONTROL-OPENCODE-CHECKPOINT-CAUSE-04 | **Source repaired on PR #2284; hosted exact-head acceptance pending** | Exact `3f05c1fc` proves two causal-context failures: provider-like words in assistant prose selected a trusted provider termination label, and an invalid control result reached the continuation as generic `nonzero-exit` although the host already held wrapper status `3`. GREEN `f57319d1` restricts classification to structured OpenCode `type=error` events, CLI stderr, and fixed host hints; GREEN `9dc05e7d` maps wrapper status `3` to `invalid-control-output`. Exact-source assertions confirm assistant export is no longer causal while export presence remains checked, and the runner continuation contract requires `Termination reason: invalid-control`. | ContextualWisdomLab/.github owns checkpoint cause identity. Keep #2284 Draft until exact-head Python/runner/security gates are terminal GREEN and an independent review qualifies; no assistant prose, prior head, or predecessor check may author the cause. | +| CONTROL-OPENCODE-CHECKPOINT-BOUNDS-05 | **Source repaired on PR #2284; hosted exact-head acceptance pending** | RED `c8680c56` executes a valid provider session export above the existing 2 MiB OpenCode evidence bound and proves that both checkpoint summary and required-output extraction consumed it in full. GREEN `5c49d4f8` introduces `MAX_SESSION_EXPORT_BYTES`, reads at most bound + 1 byte, rejects oversized or non-UTF-8 exports before JSON parsing, and replaces both unbounded `read_text()` paths. Exact remote Python compilation, the oversized/small-export behavior probe, and 31 directly executable checkpoint cases pass; three fixture-dependent cases were not represented as hosted evidence. | ContextualWisdomLab/.github owns the review-host memory boundary. Keep #2284 Draft until exact-head hosted quality/security gates and qualifying independent review are terminal; oversized provider artifacts cannot become continuation evidence. | ### 2026-09-13 current-head incident delta From 887ad44f89e1473debf3c797bec50fede6db984e Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sun, 20 Sep 2026 07:51:29 +0900 Subject: [PATCH 43/43] style(gap): preserve module and EOF spacing --- tests/test_product_technical_gap_baseline.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_product_technical_gap_baseline.py b/tests/test_product_technical_gap_baseline.py index 2ecc29481c..03eeae673d 100644 --- a/tests/test_product_technical_gap_baseline.py +++ b/tests/test_product_technical_gap_baseline.py @@ -99,6 +99,7 @@ def test_master_context_points_at_live_baseline_without_freezing_shas() -> None: assert "Done" in source assert "merge authorization" in source + def test_baseline_preserves_protected_main_authority_sections() -> None: """Partial-file replacements must not erase protected Gap evidence.""" @@ -112,4 +113,3 @@ def test_baseline_preserves_protected_main_authority_sections() -> None: "## Noema reviewer credential-lifetime delta — 2026-09-01", ): assert marker in source, marker -