diff --git a/CHANGELOG.d/20260926-launcher-free-now-cost-signal.md b/CHANGELOG.d/20260926-launcher-free-now-cost-signal.md new file mode 100644 index 0000000000..00d86a2df9 --- /dev/null +++ b/CHANGELOG.d/20260926-launcher-free-now-cost-signal.md @@ -0,0 +1,50 @@ +### Review sidecar admits free routes only when servable free now + +- `scripts/ci/contextual_orchestrator_review_launcher.py` no longer treats a + zero catalog price alone as `orchestrator/free` admission. Candidates are + the orchestrator's zero-priced rows plus rows it nominated from Experiential + Labs' public keyless `promotions[]` catalog (`free_promotion`), limited to + general-chat, text-output routes before any probe. `_free_now_models` + consults the pinned orchestrator's `free_serving_evidence` signal (the same + predicate its `general_free_serving_candidates` and + `TaskOrchestrator._is_free_agent` use): `probe_free_candidates` sends at + most `REVIEW_FREE_EVIDENCE_MAX_PROBES = 4` 16-token probes per run to + evidence-required routes (Experiential Labs), and a route is admitted only + when a response reported `usage.cost == 0` with `is_byok: false`. A positive + cost, a 429 `free_limit_reached` / `insufficient_quota`, a missing or + unparseable cost, or a failed probe keeps the route out of the free pool (it + stays eligible as priced). That a free-tier call reports exactly + `cost: 0` with `is_byok: false` is inferred from the provider docs and not + yet observed on a live response. +- Such rows reach the policy with `free_evidence: "per_call_zero_cost"` and no + token prices; `contextual_orchestrator_review_policy.parse_discovery_report` + classifies them free with `non_token_price_evidence` + `{"source": "usage.cost", "price": 0.0, "unit": "per_call"}`, only for + `experiential_labs` rows marked `is_free: true`. +- When the pinned orchestrator predates that module (or lacks + `probe_free_candidates`), those providers are treated as paid + (fail-closed), matching the pin's own serving selector so no dead route + occupies a free slot. The discovery artifact records the `free_now` signal, + probe count, probed routes, and withheld routes with reasons. +- A probe sent after the organization's free allowance is used up is + billed; this happens only when Experiential Labs' org-wide credits overflow + is on (it turns on automatically at the first real payment; otherwise the + provider answers 429 and nothing is billed). The orchestrator's + `FREE_SERVING_LEDGER` is in-process memory and every launcher run is a new + process, so the exposure is one billed call per nominated route per sidecar + run (OpenCode, Noema and Strix sidecars; at most 4 probes per run; today + only `jev-latest` is nominated). A billed response demotes the route only + for the rest of that run; the orchestrator's allowance reset is 00:05 UTC + (00:00 UTC plus a 5-minute clock-skew margin). This is an accepted + trade-off (the repository owner's decision). Runs with + `--require-zdr` send no probe at all, since Experiential Labs has no ZDR + scope and its routes would be dropped by the ZDR filter anyway; the + artifact records `probe_skipped: "require_zdr"`. Probes stay bounded by + count only; per ADR 0003 they carry no fixed wall-clock timeout. +- An unexpected shape of the pinned signal (a missing `FREE_SERVING_LEDGER`, + a changed signature, any other error from it) no longer aborts the + sidecar: only evidence-required and promotion-only routes are withheld as + `no_signal`, and the artifact records `signal: "incompatible"` with the + error type. The evidence-required provider set and the per-call marker are + defined once in `contextual_orchestrator_review_policy` and reused by the + launcher. diff --git a/scripts/ci/contextual_orchestrator_review_launcher.py b/scripts/ci/contextual_orchestrator_review_launcher.py index 373e57eee2..a765778e38 100644 --- a/scripts/ci/contextual_orchestrator_review_launcher.py +++ b/scripts/ci/contextual_orchestrator_review_launcher.py @@ -36,6 +36,8 @@ from scripts.ci.contextual_orchestrator_review_policy import ( FREE_POOL_CREDENTIAL_NAMES, + PER_CALL_COST_EVIDENCE_PROVIDERS, + PER_CALL_FREE_EVIDENCE, provider_account, ) @@ -144,6 +146,38 @@ # ranking (higher priority first; catalog priorities are 0..-11) never places a # deferred route ahead of a ready one. REVIEW_PREFLIGHT_DEFERRED_PRIORITY_PENALTY = 1000 +# Providers whose free status the pinned orchestrator can only prove from +# per-call cost evidence (``contextual_orchestrator.free_serving_evidence``). +# One definition shared with the policy's per-call marker check. Used on its +# own only when the pinned orchestrator predates (or exposes an unexpected +# shape of) that module: without the signal those rows are treated as paid +# (fail-closed), matching the pinned ``TaskOrchestrator._is_free_agent`` so no +# dead route occupies a free slot. +FREE_EVIDENCE_REQUIRED_PROVIDERS_FALLBACK = PER_CALL_COST_EVIDENCE_PROVIDERS +# At most this many one-shot cost-evidence probes per run. Each probe is the +# same 16-token plain-chat request as the preflight. +# +# Accepted trade-off (the repository owner's explicit decision): a probe sent +# after the organization's free allowance is used up is billed. That happens +# only when Experiential Labs' org-wide "credits overflow" is on (it turns on +# automatically at the organization's first real payment); otherwise the +# provider answers 429 ``free_limit_reached`` and nothing is billed. The +# orchestrator's ``FREE_SERVING_LEDGER`` lives in one process's memory and +# every launcher run is a new process, so nothing carries a demotion from one +# run to the next: the exposure is one billed call per nominated route per +# sidecar run (OpenCode, Noema and Strix sidecars), at most this many probes +# per run (today only ``jev-latest`` is nominated). Within a run, the billed +# response reports ``usage.cost > 0``, which demotes the route. The +# orchestrator holds a demotion until the allowance reset at 00:05 UTC +# (00:00 UTC plus its 5-minute ``ALLOWANCE_RESET_SKEW_SECONDS``), but the +# ledger ends with the process. Runs with ``--require-zdr`` never probe +# (Experiential Labs has no ZDR scope, so its routes would be dropped by the +# ZDR filter anyway). +# +# Probes are bounded by this count, not by a wall-clock timeout: ADR 0003 keeps +# inference, preflight, and DNS/TLS setup free of fixed timeouts, and the +# probe uses the same timeout-free ModelClient settings as the preflight. +REVIEW_FREE_EVIDENCE_MAX_PROBES = 4 class ReviewPreflightError(RuntimeError): @@ -220,8 +254,138 @@ def _route_identity(model: object) -> tuple[str, str]: ) +def _nominated_free_models(routable: list[object], free_models: list[object]) -> list[object]: + """Return zero-priced rows plus provider-promotion nominees, deduplicated. + + ``free_models`` is the pinned orchestrator's ``free_discovered_models`` + (zero token price). A row with ``free_promotion`` (set by the pinned + orchestrator from Experiential Labs' public ``promotions[]`` catalog) is a + candidate too; it is admitted only by per-call cost evidence. + """ + nominated = list(free_models) + seen = {_route_identity(model) for model in nominated} + for model in routable: + identity = _route_identity(model) + if getattr(model, "free_promotion", False) is True and identity not in seen: + nominated.append(model) + seen.add(identity) + return nominated + + +def _evidence_required_providers(evidence: Any | None) -> frozenset[str]: + """Return the providers whose free status needs per-call cost evidence.""" + if evidence is None: + return FREE_EVIDENCE_REQUIRED_PROVIDERS_FALLBACK + return frozenset(getattr(evidence, "COST_EVIDENCE_REQUIRED_PROVIDERS", ())) | ( + FREE_EVIDENCE_REQUIRED_PROVIDERS_FALLBACK + ) + + +def _withhold_evidence_required( + free_models: list[object], +) -> tuple[list[object], list[dict[str, str]]]: + """Fail-closed split without a usable signal: withhold evidence-required rows.""" + required = _evidence_required_providers(None) + kept: list[object] = [] + withheld: list[dict[str, str]] = [] + for model in free_models: + provider, model_id = _route_identity(model) + if provider in required or not getattr(model, "is_free", False): + withheld.append({"provider": provider, "model": model_id, "reason": "no_signal"}) + continue + kept.append(model) + return kept, withheld + + +def _free_now_models( + free_models: list[object], + *, + evidence: Any | None, + probe: Callable[[object], object] | None, + probe_skip_reason: str | None = None, +) -> tuple[list[object], dict[str, object]]: + """Keep only nominated routes the orchestrator can serve free *right now*. + + A zero catalog price or a provider free promotion nominates a route; it + does not prove the next call is free (Experiential Labs bills past a + per-org free allowance once credits overflow is on). ``evidence`` is the + pinned orchestrator's ``free_serving_evidence`` module -- the same signal + its discovery selector and ``TaskOrchestrator._is_free_agent`` consult -- + or ``None`` when the pin predates it. ``probe`` sends one bounded + plain-chat request; the pinned ``ModelClient`` records that response's + reported ``usage.cost`` (or ``EXHAUSTED`` for a 429 free-quota error) in + the shared ledger. + + * Signal unavailable (or a pin without ``probe_free_candidates``): + evidence-required providers are withheld (paid, fail-closed). + * Signal available: ``evidence.probe_free_candidates`` sends at most + ``REVIEW_FREE_EVIDENCE_MAX_PROBES`` probes to nominated, probe-due + evidence-required routes. Every route is then admitted only through + ``evidence.free_serving_admitted``, so a route demoted by a positive + cost or an exhausted allowance is dropped too. + * ``probe_skip_reason`` set (``"require_zdr"``): no probe is sent; + admission still reads the ledger, so evidence-required routes without + a recorded ``FREE`` verdict are withheld. + * Unexpected pin shape (a missing ``FREE_SERVING_LEDGER``, a changed + signature, or any other error from the signal): the run continues with + only evidence-required (and promotion-only) routes withheld as + ``no_signal``; the sidecar is never aborted by the free-now signal. + + Returns: + The admitted models and a secret-free report for the discovery artifact. + """ + compatible = evidence is not None and callable( + getattr(evidence, "probe_free_candidates", None) + ) + report: dict[str, object] = { + "signal": "free_serving_evidence" if compatible else "unavailable", + "probes": 0, + "probed": [], + "probe_skipped": probe_skip_reason, + "withheld": [], + } + if not compatible: + kept, report["withheld"] = _withhold_evidence_required(free_models) + return kept, report + + try: + if probe is not None and probe_skip_reason is None: + probe_report = evidence.probe_free_candidates( + free_models, probe=probe, max_probes=REVIEW_FREE_EVIDENCE_MAX_PROBES + ) + report["probes"] = int(probe_report.get("probes", 0)) + report["probed"] = [str(route) for route in probe_report.get("probed", [])] + ledger = evidence.FREE_SERVING_LEDGER + kept = [] + withheld: list[dict[str, str]] = [] + for model in free_models: + provider, model_id = _route_identity(model) + if evidence.free_serving_admitted( + provider, model_id, catalog_free=bool(getattr(model, "is_free", False)) + ): + kept.append(model) + continue + verdict = ledger.verdict(provider, model_id) + withheld.append( + { + "provider": provider, + "model": model_id, + "reason": f"cost_{verdict.value}" if verdict is not None else "no_evidence", + } + ) + except Exception as exc: # noqa: BLE001 - an unexpected pin shape must not abort the sidecar + report["signal"] = "incompatible" + report["signal_error"] = type(exc).__name__ + kept, report["withheld"] = _withhold_evidence_required(free_models) + return kept, report + report["withheld"] = withheld + return kept, report + + def _report_rows( - discovered: list[object], free_route_identities: frozenset[tuple[str, str]] + discovered: list[object], + free_route_identities: frozenset[tuple[str, str]], + per_call_free_identities: frozenset[tuple[str, str]] = frozenset(), ) -> list[dict[str, object]]: """Convert in-process discovered models into price-evidenced report rows. @@ -231,9 +395,17 @@ def _report_rows( read from the discovered model when present and otherwise falls back to the org ZDR policy table (``scripts/ci/zdr_policy.py``). + A route in ``per_call_free_identities`` was admitted by a recorded + per-call ``usage.cost == 0`` verdict rather than a zero token price (for + example an Experiential Labs promotion on a list-priced model). Its row + carries ``free_evidence: "per_call_zero_cost"`` and no token prices, so + the policy classifies it free on that evidence without reinterpreting a + list price. + Args: discovered: Selected ``discover_all_models()`` result. - free_route_identities: Routes the orchestrator attested as zero-priced. + free_route_identities: Routes the orchestrator attested as free now. + per_call_free_identities: Free routes whose evidence is per-call cost. Returns: Price-evidenced rows shaped for @@ -254,20 +426,26 @@ def _report_rows( auth_scheme = str( getattr(model, "auth_scheme", None) or zdr_policy.PROVIDER_AUTH_SCHEMES[provider] ) - rows.append( - { - "provider": provider, - "model": model_id, - "agent_id": str(getattr(model, "agent_id", None) or f"{provider}_{model_id}"), - "is_free": (provider, model_id) in free_route_identities, - "prompt_price_per_1k": getattr(model, "prompt_price_per_1k", None), - "completion_price_per_1k": getattr(model, "completion_price_per_1k", None), - "currency_code": getattr(model, "currency_code", None), - "base_url": base_url, - "credential_key": credential_key, - "auth_scheme": auth_scheme, - } - ) + per_call_free = (provider, model_id) in per_call_free_identities + row: dict[str, object] = { + "provider": provider, + "model": model_id, + "agent_id": str(getattr(model, "agent_id", None) or f"{provider}_{model_id}"), + "is_free": (provider, model_id) in free_route_identities, + "prompt_price_per_1k": None + if per_call_free + else getattr(model, "prompt_price_per_1k", None), + "completion_price_per_1k": None + if per_call_free + else getattr(model, "completion_price_per_1k", None), + "currency_code": None if per_call_free else getattr(model, "currency_code", None), + "base_url": base_url, + "credential_key": credential_key, + "auth_scheme": auth_scheme, + } + if per_call_free: + row["free_evidence"] = PER_CALL_FREE_EVIDENCE + rows.append(row) return rows @@ -1080,7 +1258,11 @@ def main(argv: list[str] | None = None) -> int: from contextual_orchestrator.credentials import get_credential from contextual_orchestrator.chat_capability import is_general_chat_agent_model_id - from contextual_orchestrator.model_discovery import discover_all_models, free_discovered_models + from contextual_orchestrator.model_discovery import ( + agent_from_discovered, + discover_all_models, + free_discovered_models, + ) from contextual_orchestrator.orchestrator import ModelClient, TaskOrchestrator, load_agents from contextual_orchestrator.review_gateway import ( REVIEW_AUTH_CREDENTIAL_NAME, @@ -1128,8 +1310,63 @@ def main(argv: list[str] | None = None) -> int: raise SystemExit(f"review sidecar discovery failed: {exc}") from exc _log_discovery_errors(discovery_errors) routable_discovered = _routable_discovered_models(discovered) - free_models = list(free_discovered_models(routable_discovered)) if routable_discovered else [] + # Only general-chat, text-output nominees can serve review traffic; filter + # before probing so no cost probe is spent on a route that is never selected. + free_models = [ + model + for model in _nominated_free_models( + routable_discovered, + list(free_discovered_models(routable_discovered)) if routable_discovered else [], + ) + if is_general_chat_agent_model_id(getattr(model, "model_id", "")) + and _has_text_output(model) + ] + try: + from contextual_orchestrator import free_serving_evidence + except ImportError: # pinned orchestrator predates the per-call cost signal + free_serving_evidence = None + evidence_client = ModelClient( + max_output_tokens=REVIEW_PREFLIGHT_BASE_TOKENS, + max_retries=0, + temperature=REVIEW_TEMPERATURE, + ) + + def _probe_free_evidence(model: object) -> object: + return evidence_client.proxy_send_once( + agent_from_discovered(model), + "chat/completions", + { + "model": getattr(model, "model_id", ""), + "messages": [{"role": "user", "content": "Reply with just 'OK'."}], + "temperature": REVIEW_TEMPERATURE, + "max_tokens": REVIEW_PREFLIGHT_BASE_TOKENS, + "stream": False, + }, + ) + + # Private-repository runs (--require-zdr) keep only ZDR-attested routes; + # evidence-required providers have no ZDR scope, so probing them would + # spend requests (possibly one billed call) on routes that are dropped. + # Note: runtime preflight later calls every admitted route once more. + free_models, free_evidence_report = _free_now_models( + free_models, + evidence=free_serving_evidence, + probe=_probe_free_evidence, + probe_skip_reason="require_zdr" if args.require_zdr else None, + ) + print( + "free_now_signal " + f"signal={free_evidence_report['signal']} probes={free_evidence_report['probes']} " + f"probe_skipped={free_evidence_report['probe_skipped'] or 'no'} " + f"admitted={len(free_models)} withheld={len(free_evidence_report['withheld'])}", + file=sys.stderr, + flush=True, + ) free_route_identities = frozenset(_route_identity(model) for model in free_models) + evidence_required = _evidence_required_providers(free_serving_evidence) + per_call_free_identities = frozenset( + identity for identity in free_route_identities if identity[0] in evidence_required + ) selected_models = [] for model in routable_discovered: model_id = getattr(model, "model_id", "") @@ -1143,8 +1380,8 @@ def main(argv: list[str] | None = None) -> int: f"review sidecar discovered no eligible models; orchestrator/{args.pool} would fail closed" ) - rows = _report_rows(selected_models, free_route_identities) - _write_json(args.discovery_out, {"models": rows}) + rows = _report_rows(selected_models, free_route_identities, per_call_free_identities) + _write_json(args.discovery_out, {"models": rows, "free_now": free_evidence_report}) zdr_endpoints = _load_zdr_endpoints(args.zdr_endpoints) normalized_rows = parse_discovery_report({"models": rows}) free_rows = [ diff --git a/scripts/ci/contextual_orchestrator_review_policy.py b/scripts/ci/contextual_orchestrator_review_policy.py index 9c204640d2..cd324742ec 100644 --- a/scripts/ci/contextual_orchestrator_review_policy.py +++ b/scripts/ci/contextual_orchestrator_review_policy.py @@ -172,6 +172,34 @@ def _bytez_non_token_price_evidence( return None +PER_CALL_FREE_EVIDENCE = "per_call_zero_cost" +"""Launcher marker: route admitted free by a recorded per-call ``usage.cost == 0``.""" + +PER_CALL_COST_EVIDENCE_PROVIDERS = frozenset({"experiential_labs"}) +"""Providers whose free status is proven by per-call cost, not token prices.""" + + +def _per_call_cost_price_evidence( + *, provider: str, is_free: bool, free_evidence: object +) -> dict[str, object] | None: + """Accept per-call zero-cost evidence for an evidence-required provider. + + The launcher admits these routes only after the pinned orchestrator + recorded a ``usage.cost == 0`` (with ``is_byok: false``) verdict for the + route; a promotional free call on a list-priced model has no zero token + price to show, so token prices are dropped instead of reinterpreted. The + marker is honoured only for providers in + :data:`PER_CALL_COST_EVIDENCE_PROVIDERS` and only with ``is_free: true``. + """ + if ( + is_free + and free_evidence == PER_CALL_FREE_EVIDENCE + and provider in PER_CALL_COST_EVIDENCE_PROVIDERS + ): + return {"source": "usage.cost", "price": 0.0, "unit": "per_call"} + return None + + def parse_discovery_report(report: Mapping[str, Any]) -> list[dict[str, Any]]: """Validate and normalize a contextual-orchestrator discovery report.""" rows = report.get("models") @@ -223,7 +251,11 @@ def parse_discovery_report(report: Mapping[str, Any]) -> list[dict[str, Any]]: completion_price=completion_price_input, ) if provider == "bytez" - else None + else _per_call_cost_price_evidence( + provider=provider, + is_free=is_free, + free_evidence=row.get("free_evidence"), + ) ) if non_token_price_evidence is not None: cost_evidence = COST_FREE diff --git a/tests/test_contextual_orchestrator_free_now_signal.py b/tests/test_contextual_orchestrator_free_now_signal.py new file mode 100644 index 0000000000..e13232c9ad --- /dev/null +++ b/tests/test_contextual_orchestrator_free_now_signal.py @@ -0,0 +1,518 @@ +"""The launcher admits nominated free routes only when they are servable free now.""" + +from __future__ import annotations + +import contextlib +import enum +from types import SimpleNamespace + +import pytest + +from scripts.ci.contextual_orchestrator_review_launcher import ( + FREE_EVIDENCE_REQUIRED_PROVIDERS_FALLBACK, + PER_CALL_FREE_EVIDENCE, + REVIEW_FREE_EVIDENCE_MAX_PROBES, + _free_now_models, + _nominated_free_models, + _report_rows, +) +from scripts.ci.contextual_orchestrator_review_policy import ( + COST_FREE, + COST_PRICED, + PolicyError, + parse_discovery_report, +) + + +class _Verdict(str, enum.Enum): + FREE = "free" + PAID = "paid" + EXHAUSTED = "exhausted" + UNKNOWN = "unknown" + + +class _Ledger: + def __init__(self) -> None: + self.verdicts: dict[tuple[str, str], _Verdict] = {} + + def record(self, provider: str, model: str, verdict: _Verdict) -> None: + self.verdicts[(provider, model)] = verdict + + def verdict(self, provider: str, model: str) -> _Verdict | None: + return self.verdicts.get((provider, model)) + + def probe_due(self, provider: str, model: str) -> bool: + # The real ledger also re-opens non-FREE routes after 00:00 UTC; the + # launcher runs in a fresh process, so "never observed" is the case here. + return (provider, model) not in self.verdicts + + +def _evidence_module() -> SimpleNamespace: + """Mirror contextual_orchestrator.free_serving_evidence's public contract.""" + ledger = _Ledger() + required = frozenset({"experiential_labs"}) + + def admitted(provider, model, *, catalog_free): + verdict = ledger.verdict(provider, model) + if provider in required: + return verdict is _Verdict.FREE + return bool(catalog_free) and verdict not in {_Verdict.PAID, _Verdict.EXHAUSTED} + + def probe_free_candidates(models, *, probe, max_probes): + probed = [] + for model in models: + if len(probed) >= max_probes: + break + key = (model.provider_name, model.model_id) + nominated = getattr(model, "is_free", False) or getattr(model, "free_promotion", False) + if key[0] not in required or not nominated or not ledger.probe_due(*key): + continue + probed.append("/".join(key)) + with contextlib.suppress(Exception): # mirrors the real module + probe(model) + if ledger.verdict(*key) is None: + ledger.record(*key, _Verdict.UNKNOWN) + return {"probes": len(probed), "probed": probed} + + return SimpleNamespace( + COST_EVIDENCE_REQUIRED_PROVIDERS=required, + FREE_SERVING_LEDGER=ledger, + CostVerdict=_Verdict, + free_serving_admitted=admitted, + probe_free_candidates=probe_free_candidates, + ) + + +def _model( + provider: str, model_id: str, *, is_free: bool = True, free_promotion: bool = False +) -> SimpleNamespace: + return SimpleNamespace( + provider_name=provider, + model_id=model_id, + is_free=is_free, + free_promotion=free_promotion, + prompt_price_per_1k=0.0 if is_free else 0.5, + completion_price_per_1k=0.0 if is_free else 1.5, + currency_code="USD", + ) + + +def test_without_the_orchestrator_signal_experiential_is_treated_as_paid() -> None: + """Fail closed when the pinned orchestrator has no per-call cost signal.""" + models = [ + _model("experiential_labs", "promo"), + _model("experiential_labs", "promoted", is_free=False, free_promotion=True), + _model("openrouter", "m:free"), + ] + + kept, report = _free_now_models(models, evidence=None, probe=None) + + assert [m.provider_name for m in kept] == ["openrouter"] + assert "experiential_labs" in FREE_EVIDENCE_REQUIRED_PROVIDERS_FALLBACK + assert report == { + "signal": "unavailable", + "probes": 0, + "probed": [], + "probe_skipped": None, + "withheld": [ + {"provider": "experiential_labs", "model": "promo", "reason": "no_signal"}, + {"provider": "experiential_labs", "model": "promoted", "reason": "no_signal"}, + ], + } + + +def test_a_pin_without_probe_support_is_also_fail_closed() -> None: + evidence = _evidence_module() + del evidence.probe_free_candidates + evidence.FREE_SERVING_LEDGER.record("experiential_labs", "promo", _Verdict.FREE) + + kept, report = _free_now_models( + [_model("experiential_labs", "promo"), _model("openrouter", "m:free")], + evidence=evidence, + probe=lambda _model: None, + ) + + assert [m.model_id for m in kept] == ["m:free"] + assert report["signal"] == "unavailable" + + +def test_promotion_nominees_join_zero_priced_rows_once() -> None: + zero = _model("openrouter", "m:free") + promoted = _model("experiential_labs", "promo", is_free=False, free_promotion=True) + priced = _model("experiential_labs", "paid", is_free=False) + duplicate = _model("openrouter", "m:free", free_promotion=True) + + nominated = _nominated_free_models([zero, promoted, priced, duplicate], [zero]) + + assert nominated == [zero, promoted] + + +def test_zero_cost_probe_admits_and_positive_cost_or_quota_withholds() -> None: + evidence = _evidence_module() + costs = { + "free-now": _Verdict.FREE, + "overflowing": _Verdict.PAID, + "quota-429": _Verdict.EXHAUSTED, + } + probed: list[str] = [] + + def probe(model): + probed.append(model.model_id) + if model.model_id == "transport-error": + raise RuntimeError("connection reset") + if model.model_id in costs: + evidence.FREE_SERVING_LEDGER.record( + model.provider_name, model.model_id, costs[model.model_id] + ) + + models = [ + _model("experiential_labs", "free-now", is_free=False, free_promotion=True), + _model("experiential_labs", "overflowing"), + _model("experiential_labs", "quota-429"), + _model("experiential_labs", "transport-error"), + _model("openrouter", "m:free"), + ] + kept, report = _free_now_models(models, evidence=evidence, probe=probe) + + assert [m.model_id for m in kept] == ["free-now", "m:free"] + assert probed == ["free-now", "overflowing", "quota-429", "transport-error"] + assert report["signal"] == "free_serving_evidence" + assert report["probes"] == 4 + assert report["probed"] == [f"experiential_labs/{name}" for name in probed] + assert report["withheld"] == [ + {"provider": "experiential_labs", "model": "overflowing", "reason": "cost_paid"}, + {"provider": "experiential_labs", "model": "quota-429", "reason": "cost_exhausted"}, + {"provider": "experiential_labs", "model": "transport-error", "reason": "cost_unknown"}, + ] + + +def test_probe_budget_is_bounded_and_unprobed_routes_stay_paid() -> None: + evidence = _evidence_module() + calls: list[str] = [] + + def probe(model): + calls.append(model.model_id) + evidence.FREE_SERVING_LEDGER.record(model.provider_name, model.model_id, _Verdict.FREE) + + models = [ + _model("experiential_labs", f"m{index}") + for index in range(REVIEW_FREE_EVIDENCE_MAX_PROBES + 2) + ] + kept, report = _free_now_models(models, evidence=evidence, probe=probe) + + assert len(calls) == REVIEW_FREE_EVIDENCE_MAX_PROBES + assert len(kept) == REVIEW_FREE_EVIDENCE_MAX_PROBES + assert [row["reason"] for row in report["withheld"]] == ["no_evidence", "no_evidence"] + + +def test_existing_verdicts_are_reused_and_demote_catalog_free_routes() -> None: + evidence = _evidence_module() + evidence.FREE_SERVING_LEDGER.record("experiential_labs", "known", _Verdict.FREE) + evidence.FREE_SERVING_LEDGER.record("experiential_labs", "spent", _Verdict.EXHAUSTED) + evidence.FREE_SERVING_LEDGER.record("openrouter", "m:free", _Verdict.PAID) + + def probe(model): # pragma: no cover - must not be called + raise AssertionError(f"unexpected probe for {model.model_id}") + + kept, report = _free_now_models( + [ + _model("experiential_labs", "known"), + _model("experiential_labs", "spent"), + _model("openrouter", "m:free"), + ], + evidence=evidence, + probe=probe, + ) + + assert [m.model_id for m in kept] == ["known"] + assert report["probes"] == 0 + assert report["withheld"] == [ + {"provider": "experiential_labs", "model": "spent", "reason": "cost_exhausted"}, + {"provider": "openrouter", "model": "m:free", "reason": "cost_paid"}, + ] + + +def test_per_call_free_rows_drop_list_prices_and_parse_as_free() -> None: + promoted = _model("experiential_labs", "promo", is_free=False, free_promotion=True) + zero = _model("openrouter", "m:free") + identities = frozenset({("experiential_labs", "promo"), ("openrouter", "m:free")}) + + rows = _report_rows( + [promoted, zero], identities, frozenset({("experiential_labs", "promo")}) + ) + + assert rows[0]["free_evidence"] == PER_CALL_FREE_EVIDENCE + assert rows[0]["is_free"] is True + assert ( + rows[0]["prompt_price_per_1k"], + rows[0]["completion_price_per_1k"], + rows[0]["currency_code"], + ) == (None, None, None) + assert "free_evidence" not in rows[1] + parsed = parse_discovery_report({"models": rows}) + assert [row["cost_evidence"] for row in parsed] == [COST_FREE, COST_FREE] + assert parsed[0]["non_token_price_evidence"] == { + "source": "usage.cost", + "price": 0.0, + "unit": "per_call", + } + assert parsed[1]["non_token_price_evidence"] is None + + +def _experiential_row(**overrides: object) -> dict[str, object]: + row: dict[str, object] = { + "provider": "experiential_labs", + "model": "promo", + "is_free": True, + "prompt_price_per_1k": None, + "completion_price_per_1k": None, + "currency_code": None, + "credential_key": "EXPERIENTIAL_LABS_API_KEY", + "free_evidence": PER_CALL_FREE_EVIDENCE, + } + row.update(overrides) + return row + + +def test_policy_honours_the_marker_only_for_free_evidence_required_rows() -> None: + # Without the marker a free row with no token prices stays unknown. + unmarked = parse_discovery_report({"models": [_experiential_row(free_evidence=None)]}) + assert unmarked[0]["cost_evidence"] == "unknown" + # A marker on a non-free row is ignored (list prices decide). + priced = parse_discovery_report( + { + "models": [ + _experiential_row( + is_free=False, + prompt_price_per_1k=0.5, + completion_price_per_1k=1.5, + currency_code="USD", + ) + ] + } + ) + assert priced[0]["cost_evidence"] == COST_PRICED + # A marker on a provider outside the per-call set is ignored too. + other = parse_discovery_report( + { + "models": [ + { + "provider": "openrouter", + "model": "vendor/m", + "is_free": True, + "prompt_price_per_1k": None, + "completion_price_per_1k": None, + "free_evidence": PER_CALL_FREE_EVIDENCE, + } + ] + } + ) + assert other[0]["cost_evidence"] == "unknown" + # A free marker never launders a conflicting list price. + with pytest.raises(PolicyError, match="conflicts with its free price marker"): + parse_discovery_report( + { + "models": [ + { + "provider": "openrouter", + "model": "vendor/m", + "is_free": True, + "prompt_price_per_1k": 0.5, + "completion_price_per_1k": 1.5, + "currency_code": "USD", + "free_evidence": PER_CALL_FREE_EVIDENCE, + } + ] + } + ) + + +# --- Round 3: ZDR runs never probe, unexpected pin shapes never abort ----------- + + +def test_probe_skip_reason_sends_no_probe_and_keeps_evidence_routes_paid() -> None: + evidence = _evidence_module() + + def probe(model): # pragma: no cover - must not be called + raise AssertionError(f"unexpected probe for {model.model_id}") + + kept, report = _free_now_models( + [ + _model("experiential_labs", "promo", is_free=False, free_promotion=True), + _model("openrouter", "m:free"), + ], + evidence=evidence, + probe=probe, + probe_skip_reason="require_zdr", + ) + + assert [m.model_id for m in kept] == ["m:free"] + assert report["probes"] == 0 + assert report["probe_skipped"] == "require_zdr" + assert report["withheld"] == [ + {"provider": "experiential_labs", "model": "promo", "reason": "no_evidence"} + ] + + +def _without(evidence: SimpleNamespace, name: str) -> SimpleNamespace: + delattr(evidence, name) + return evidence + + +def _old_signature(evidence: SimpleNamespace) -> SimpleNamespace: + evidence.free_serving_admitted = lambda provider, model: True # no catalog_free kwarg + return evidence + + +def _raising_probe_runner(evidence: SimpleNamespace) -> SimpleNamespace: + def probe_free_candidates(models, *, probe, max_probes): + raise AttributeError("'DiscoveredModel' object has no attribute 'provider_name'") + + evidence.probe_free_candidates = probe_free_candidates + return evidence + + +def _non_mapping_report(evidence: SimpleNamespace) -> SimpleNamespace: + evidence.probe_free_candidates = lambda models, *, probe, max_probes: ["unexpected"] + return evidence + + +@pytest.mark.parametrize( + ("mutate", "error"), + [ + (lambda evidence: _without(evidence, "FREE_SERVING_LEDGER"), "AttributeError"), + (_old_signature, "TypeError"), + (_raising_probe_runner, "AttributeError"), + (_non_mapping_report, "AttributeError"), + ], + ids=["missing_ledger", "signature_type_error", "attribute_error", "report_shape"], +) +def test_an_unexpected_pin_shape_withholds_only_evidence_routes(mutate, error) -> None: + evidence = mutate(_evidence_module()) + + kept, report = _free_now_models( + [ + _model("experiential_labs", "promo"), + _model("experiential_labs", "promoted", is_free=False, free_promotion=True), + _model("openrouter", "m:free"), + ], + evidence=evidence, + probe=lambda _model: None, + ) + + assert [m.model_id for m in kept] == ["m:free"] + assert report["signal"] == "incompatible" + assert report["signal_error"] == error + assert report["withheld"] == [ + {"provider": "experiential_labs", "model": "promo", "reason": "no_signal"}, + {"provider": "experiential_labs", "model": "promoted", "reason": "no_signal"}, + ] + + +def test_the_fallback_provider_set_is_the_policy_set() -> None: + from scripts.ci import contextual_orchestrator_review_policy as policy + + assert FREE_EVIDENCE_REQUIRED_PROVIDERS_FALLBACK is policy.PER_CALL_COST_EVIDENCE_PROVIDERS + assert PER_CALL_FREE_EVIDENCE == policy.PER_CALL_FREE_EVIDENCE + + +def _stub_orchestrator(monkeypatch, *, probe_verdict: _Verdict) -> SimpleNamespace: + """Install a minimal vendored-orchestrator stub so ``main()`` runs to selection.""" + import sys + import types + + from scripts.ci import contextual_orchestrator_review_launcher as launcher + + evidence = _evidence_module() + calls = SimpleNamespace(probe_runner=0, sent=[], clients=[]) + real_runner = evidence.probe_free_candidates + + def probe_free_candidates(models, *, probe, max_probes): + calls.probe_runner += 1 + return real_runner(models, probe=probe, max_probes=max_probes) + + evidence.probe_free_candidates = probe_free_candidates + + class _Client: + def __init__(self, *args, **kwargs) -> None: + calls.clients.append(kwargs) + + def proxy_send_once(self, agent, endpoint, payload): + calls.sent.append((agent, endpoint, payload["model"])) + evidence.FREE_SERVING_LEDGER.record("experiential_labs", payload["model"], probe_verdict) + return {} + + promo = SimpleNamespace( + provider_name="experiential_labs", + model_id="promo", + is_free=False, + free_promotion=True, + output_modalities=("text",), + prompt_price_per_1k=0.5, + completion_price_per_1k=1.5, + currency_code="USD", + ) + modules = { + "contextual_orchestrator": types.ModuleType("contextual_orchestrator"), + "contextual_orchestrator.credentials": SimpleNamespace(get_credential=lambda _name: "tok"), + "contextual_orchestrator.chat_capability": SimpleNamespace( + is_general_chat_agent_model_id=lambda _model_id: True + ), + "contextual_orchestrator.model_discovery": SimpleNamespace( + agent_from_discovered=lambda model: f"agent:{model.model_id}", + discover_all_models=lambda: ([promo], []), + free_discovered_models=lambda models: [m for m in models if m.is_free], + ), + "contextual_orchestrator.orchestrator": SimpleNamespace( + ModelClient=_Client, TaskOrchestrator=object, load_agents=lambda _path: [] + ), + "contextual_orchestrator.review_gateway": SimpleNamespace( + REVIEW_AUTH_CREDENTIAL_NAME="REVIEW_AUTH", + register_review_credentials=lambda _env: ["EXPERIENTIAL_LABS_API_KEY"], + ), + "contextual_orchestrator.server": SimpleNamespace(SecurityConfig=object, serve=None), + "contextual_orchestrator.debug_logging": SimpleNamespace(configure_logging=None), + "contextual_orchestrator.free_serving_evidence": evidence, + } + modules["contextual_orchestrator"].free_serving_evidence = evidence + for name, module in modules.items(): + monkeypatch.setitem(sys.modules, name, module) + monkeypatch.setattr(launcher, "_configure_sidecar_logging", lambda _configure: "INFO") + calls.main = launcher.main + return calls + + +def _main_args(tmp_path, *extra: str) -> list[str]: + return [ + "--discovery-out", str(tmp_path / "discovery.json"), + "--catalog-out", str(tmp_path / "catalog.json"), + "--report-out", str(tmp_path / "report.json"), + "--preflight-out", str(tmp_path / "preflight.json"), + *extra, + ] + + +def test_require_zdr_runs_never_send_a_free_evidence_probe(monkeypatch, tmp_path, capsys) -> None: + calls = _stub_orchestrator(monkeypatch, probe_verdict=_Verdict.FREE) + + with pytest.raises(SystemExit, match="no eligible models"): + calls.main(_main_args(tmp_path, "--require-zdr")) + + assert calls.probe_runner == 0 + assert calls.sent == [] + assert "probe_skipped=require_zdr" in capsys.readouterr().err + + +def test_public_runs_probe_without_a_wall_clock_timeout(monkeypatch, tmp_path, capsys) -> None: + # A billed probe (cost > 0) demotes the route, so selection still fails closed. + calls = _stub_orchestrator(monkeypatch, probe_verdict=_Verdict.PAID) + + with pytest.raises(SystemExit, match="no eligible models"): + calls.main(_main_args(tmp_path)) + + assert calls.probe_runner == 1 + assert calls.sent == [("agent:promo", "chat/completions", "promo")] + # ADR 0003: no fixed inference or connect timeout on the probe client. + assert "timeout" not in calls.clients[0] + assert "connect_timeout" not in calls.clients[0] + assert "probes=1 probe_skipped=no" in capsys.readouterr().err diff --git a/tests/test_contextual_orchestrator_review_sidecar_contract.py b/tests/test_contextual_orchestrator_review_sidecar_contract.py index 104b95f5e5..c6c1988ec6 100644 --- a/tests/test_contextual_orchestrator_review_sidecar_contract.py +++ b/tests/test_contextual_orchestrator_review_sidecar_contract.py @@ -334,9 +334,14 @@ def test_launcher_uses_orchestrator_discovery_and_governed_pools() -> None: """Discovery, price evidence, and serving come from the vendored library.""" text = _read(LAUNCHER) assert "from contextual_orchestrator.chat_capability import is_general_chat_agent_model_id" in text - assert "from contextual_orchestrator.model_discovery import discover_all_models, free_discovered_models" in text + assert "from contextual_orchestrator.model_discovery import (" in text + for name in ("agent_from_discovered", "discover_all_models", "free_discovered_models"): + assert f" {name},\n" in text assert "routable_discovered = _routable_discovered_models(discovered)" in text assert "free_discovered_models(routable_discovered)" in text + # Zero price only nominates; the orchestrator's per-call cost signal admits. + assert "from contextual_orchestrator import free_serving_evidence" in text + assert "free_models, free_evidence_report = _free_now_models(" in text assert 'getattr(model, "evidence_only", False)' in text assert 'getattr(model, "output_modalities", None)' in text assert 'isinstance(modalities, str)' in text