From d81e0fcf3aceba97ee46e628493dca40dd9b187a Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Thu, 8 Oct 2026 18:21:33 +0000 Subject: [PATCH 01/30] Add backward-compatible smart routing configuration versions --- README.md | 27 +- skills/smart-router-orchestrator/README.md | 7 +- src/ucode/cli.py | 20 +- src/ucode/constants.py | 1 + src/ucode/smart_routing/config.py | 55 ++++ src/ucode/smart_routing/orchestrator.py | 3 +- src/ucode/smart_routing/session_env.py | 3 +- src/ucode/smart_routing/v2.py | 19 +- tests/README.md | 5 + tests/conftest.py | 2 + tests/integration/README.md | 5 + tests/test_smart_routing_config.py | 363 +++++++++++++++++++++ 12 files changed, 485 insertions(+), 25 deletions(-) create mode 100644 src/ucode/smart_routing/config.py create mode 100644 tests/test_smart_routing_config.py diff --git a/README.md b/README.md index d76cc92b4..aeb923d89 100644 --- a/README.md +++ b/README.md @@ -248,15 +248,28 @@ The generated shell hooks expect Git Bash; PowerShell-only setups are not covere ### Smart Router Orchestrator -Smart-routed Claude and Codex sessions install `smart-router`. Set -`ENABLE_SMART_ROUTER_ORCHESTRATOR=1` at launch to also install and activate Smart Router -Orchestrator through the bundled `smart-router-orchestrator` skill; orchestration is off by -default. For example: +Use `SMART_ROUTING_CONFIG_VERSION` at launch to select a smart-routing configuration: + +| Version | Subagent routing | First-prompt routing | Orchestrator | +| --- | --- | --- | --- | +| `subagent_only` | On | Off | Off | +| `subagent_orch` | On | Off | On | + +Smart-routed Claude and Codex sessions install `smart-router`. The `subagent_orch` +version also installs and activates the bundled `smart-router-orchestrator` skill. +For example: ```bash -ENABLE_SMART_ROUTER_ORCHESTRATOR=1 ENABLE_SMART_ROUTING_SUBAGENT_ONLY=1 ug claude +SMART_ROUTING_CONFIG_VERSION=subagent_orch ug claude ``` +The version takes precedence over conflicting legacy flags. UG expands it into +`ENABLE_SMART_ROUTING_V2`, `ENABLE_SMART_ROUTING_SUBAGENT_ONLY`, and +`ENABLE_SMART_ROUTER_ORCHESTRATOR` for the launched session. When the version is +unset or empty, these legacy flags retain their existing behavior, including +first-prompt routing through `ENABLE_SMART_ROUTING_V2=1`. Unknown versions produce +an error listing the supported values. Orchestration remains off by default. + Use `ug codex` in the same command for Codex. Smart Router Orchestrator assigns bounded work to explorer, researcher, worker, tester, and reviewer roles while the root plans, integrates, and verifies results. Easy tasks and explicit requests not to delegate @@ -271,8 +284,8 @@ Once opted in, orchestration follows the existing smart-routing launch eligibili and session controls. Turning Smart Router off through its skill stops new automatic delegation; turning it on restores orchestration only in opted-in sessions. Explicit user requests for subagents still use normal harness behavior while routing is off. -Stored skill files do not activate orchestration when the feature flag is unset or -`ENABLE_SMART_ROUTER_ORCHESTRATOR=0`, or in non-routed sessions. Existing Isaac pilot gating +Stored skill files do not activate orchestration without an opted-in configuration, +or in non-routed sessions. Existing Isaac pilot gating and UG launch exclusions still apply. Hooks refresh orchestration state before each prompt and after compaction. A diff --git a/skills/smart-router-orchestrator/README.md b/skills/smart-router-orchestrator/README.md index 9e5aad4ed..2fb507f91 100644 --- a/skills/smart-router-orchestrator/README.md +++ b/skills/smart-router-orchestrator/README.md @@ -1,8 +1,11 @@ # Smart Router Orchestrator UG bundles the `smart-router-orchestrator` workflow and five Claude role definitions. -Smart-routed Claude and Codex launches install and -activate this skill alongside `smart-router` only with `ENABLE_SMART_ROUTER_ORCHESTRATOR=1`. +Smart-routed Claude and Codex launches install and activate this skill alongside +`smart-router` with `SMART_ROUTING_CONFIG_VERSION=subagent_orch`. +UG expands that version into the session's legacy feature flags. +The existing `ENABLE_SMART_ROUTER_ORCHESTRATOR=1` opt-in remains supported when +`SMART_ROUTING_CONFIG_VERSION` is unset; `subagent_only` explicitly leaves orchestration off. The feature is off by default; routing alone installs only `smart-router`. The workflow is injected before root prompts and after compaction. The hook checks diff --git a/src/ucode/cli.py b/src/ucode/cli.py index 4aa7e1f6e..ed49c1c13 100644 --- a/src/ucode/cli.py +++ b/src/ucode/cli.py @@ -2287,10 +2287,15 @@ def _auto_configure_tool(tool: str, custom_oauth: CustomOAuthConfig | None = Non @contextmanager def _smart_routing_v2_flag(enabled: bool | None) -> Iterator[None]: """Apply an explicit routing choice without leaking into an embedding process.""" - if enabled is None: - yield - return - previous = smart_routing_v2.override_smart_routing(enabled) + try: + previous = ( + smart_routing_v2.apply_config() + if enabled is None + else smart_routing_v2.override_smart_routing(enabled) + ) + except RuntimeError as exc: + print_err(str(exc)) + raise typer.Exit(1) from None try: yield finally: @@ -3031,9 +3036,10 @@ def default( return set_dry_run(dry_run) try: - _launch_managed_default( - ctx, dry_run=dry_run, skip_preflight=skip_preflight, workspace=workspace - ) + with _smart_routing_v2_flag(None): + _launch_managed_default( + ctx, dry_run=dry_run, skip_preflight=skip_preflight, workspace=workspace + ) except typer.Exit: # `typer.Exit` subclasses RuntimeError, so it has to be re-raised ahead of the handler # below. Otherwise a launch that already reported its own error is followed by diff --git a/src/ucode/constants.py b/src/ucode/constants.py index 38fe7e774..590f94f33 100644 --- a/src/ucode/constants.py +++ b/src/ucode/constants.py @@ -6,6 +6,7 @@ ENABLE_SMART_ROUTING_ENV_VAR = "ENABLE_SMART_ROUTING_V2" ENABLE_SUBAGENT_ROUTING_ENV_VAR = "ENABLE_SMART_ROUTING_SUBAGENT_ONLY" ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR = "ENABLE_SMART_ROUTER_ORCHESTRATOR" +SMART_ROUTING_CONFIG_VERSION_ENV_VAR = "SMART_ROUTING_CONFIG_VERSION" SMART_ROUTING_ENV_KEYS = ( ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, diff --git a/src/ucode/smart_routing/config.py b/src/ucode/smart_routing/config.py new file mode 100644 index 000000000..8c71f6fa3 --- /dev/null +++ b/src/ucode/smart_routing/config.py @@ -0,0 +1,55 @@ +"""Resolve external smart-routing versions into backward-compatible feature flags.""" + +from __future__ import annotations + +import os +from collections.abc import Mapping, MutableMapping + +from ucode.constants import ( + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, + ENABLE_SMART_ROUTING_ENV_VAR, + ENABLE_SUBAGENT_ROUTING_ENV_VAR, + SMART_ROUTING_CONFIG_VERSION_ENV_VAR, +) + +_VERSIONS = { + "subagent_only": { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + }, + "subagent_orch": { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + }, +} + + +def resolve_environment(env: Mapping[str, str] | None = None) -> dict[str, str]: + """Expand a version before applying any launch or session-specific overrides.""" + resolved = dict(os.environ if env is None else env) + version = resolved.pop(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "").strip() + if not version: + return resolved + if version not in _VERSIONS: + raise RuntimeError( + f"Unknown {SMART_ROUTING_CONFIG_VERSION_ENV_VAR} value {version!r}. " + f"Use one of: {', '.join(_VERSIONS)}, or unset it to use the legacy flags." + ) + resolved.update(_VERSIONS[version]) + return resolved + + +def apply_config(env: MutableMapping[str, str] | None = None) -> dict[str, str | None]: + """Consume the launch selector, returning the values needed to restore its input.""" + target = os.environ if env is None else env + version = target.get(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "").strip() + if not version: + return {} + resolved = resolve_environment(target) + keys = (*_VERSIONS[version], SMART_ROUTING_CONFIG_VERSION_ENV_VAR) + previous = {key: target.get(key) for key in keys} + target.update({key: resolved[key] for key in _VERSIONS[version]}) + target.pop(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, None) + return previous diff --git a/src/ucode/smart_routing/orchestrator.py b/src/ucode/smart_routing/orchestrator.py index c86d4b0a5..61d9681ed 100644 --- a/src/ucode/smart_routing/orchestrator.py +++ b/src/ucode/smart_routing/orchestrator.py @@ -14,6 +14,7 @@ from ucode import skills from ucode.constants import ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR +from ucode.smart_routing.config import resolve_environment from ucode.smart_routing.hooks import sync_managed_hooks from ucode.smart_routing.session_env import effective_environment, session_env_path @@ -27,7 +28,7 @@ def feature_enabled(env: Mapping[str, str] | None = None) -> bool: - source = os.environ if env is None else env + source = resolve_environment(env) return source.get(ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR) == "1" diff --git a/src/ucode/smart_routing/session_env.py b/src/ucode/smart_routing/session_env.py index 21887c945..d5f4a467a 100644 --- a/src/ucode/smart_routing/session_env.py +++ b/src/ucode/smart_routing/session_env.py @@ -11,6 +11,7 @@ from ucode.config_io import atomic_write_json from ucode.constants import SMART_ROUTING_ENV_KEYS +from ucode.smart_routing.config import resolve_environment SESSION_ENV_VAR = "UCODE_SESSION_ENV_FILE" SESSION_PYTHON_ENV_VAR = "UCODE_SMART_ROUTER_PYTHON" @@ -53,7 +54,7 @@ def _read(path: Path) -> dict[str, str]: def effective_environment(env: Mapping[str, str] | None = None) -> dict[str, str]: """Overlay the latest session controls on the hook process environment.""" - effective = dict(os.environ if env is None else env) + effective = resolve_environment(env) try: path = session_env_path(effective) except RuntimeError: diff --git a/src/ucode/smart_routing/v2.py b/src/ucode/smart_routing/v2.py index 5e4b1879b..4e334849e 100644 --- a/src/ucode/smart_routing/v2.py +++ b/src/ucode/smart_routing/v2.py @@ -31,6 +31,7 @@ ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, LOOPBACK_HOST, + SMART_ROUTING_CONFIG_VERSION_ENV_VAR, SMART_ROUTING_ENV_KEYS, ) from ucode.custom_oauth import custom_oauth_cli_enabled, get_custom_client_token @@ -55,6 +56,7 @@ sync_smart_routing_hooks, ) from ucode.smart_routing.codex_hooks import merge_pre_tool_use_hooks, routing_models +from ucode.smart_routing.config import apply_config, resolve_environment from ucode.smart_routing.session_env import SESSION_ENV_VAR, SESSION_PYTHON_ENV_VAR, start_session from ucode.ui import print_warning @@ -152,7 +154,7 @@ def _model_picker_catalog() -> AnthropicModelCatalog | None: def smart_routing_enabled( env: MutableMapping[str, str] | None = None, *, default: bool = False ) -> bool: - source = os.environ if env is None else env + source = resolve_environment(env) values = [source.get(var) for var in SMART_ROUTING_ENV_KEYS] if "1" in values: return True @@ -163,7 +165,7 @@ def smart_routing_enabled( def first_prompt_routing_enabled(env: MutableMapping[str, str] | None = None) -> bool: """Whether the first prompt is routed. Subagent-only wins over the full V2 flag.""" - source = os.environ if env is None else env + source = resolve_environment(env) return ( source.get(ENABLE_SMART_ROUTING_ENV_VAR) == "1" and source.get(ENABLE_SUBAGENT_ROUTING_ENV_VAR) != "1" @@ -174,10 +176,7 @@ def enable_smart_routing( env: MutableMapping[str, str] | None = None, ) -> dict[str, str | None]: """Set the full smart-routing env var and return the prior value of every routing var.""" - target = os.environ if env is None else env - previous = {var: target.get(var) for var in SMART_ROUTING_ENV_KEYS} - target[ENABLE_SMART_ROUTING_ENV_VAR] = "1" - return previous + return override_smart_routing(True, env) def override_smart_routing( @@ -187,6 +186,7 @@ def override_smart_routing( """Set an explicit launch-scoped routing choice and return the prior values.""" target = os.environ if env is None else env previous = {var: target.get(var) for var in SMART_ROUTING_ENV_KEYS} + previous.update(apply_config(target)) if enabled: target[ENABLE_SMART_ROUTING_ENV_VAR] = "1" else: @@ -211,7 +211,12 @@ def disable_smart_routing( ) -> dict[str, str | None]: """Temporarily remove the smart-routing env vars and return their prior values.""" target = os.environ if env is None else env - return {var: target.pop(var, None) for var in SMART_ROUTING_ENV_KEYS} + previous = {var: target.pop(var, None) for var in SMART_ROUTING_ENV_KEYS} + if SMART_ROUTING_CONFIG_VERSION_ENV_VAR in target: + previous[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] = target.pop( + SMART_ROUTING_CONFIG_VERSION_ENV_VAR + ) + return previous def _loopback_websocket_url(port: int) -> str: diff --git a/tests/README.md b/tests/README.md index f039e7d4b..bb7500cfd 100644 --- a/tests/README.md +++ b/tests/README.md @@ -135,6 +135,11 @@ that Claude settings and Codex's shell policy carry the interpreter and session These are component checks; they do not establish native skill permission matching or PowerShell execution. +`test_smart_routing_config.py` covers versioned selector resolution, legacy-flag precedence, +environment materialization/restoration, native-subcommand suppression, and orchestrator +state transitions through real temporary session files. These are component checks; they do +not establish live agent, hook, or gateway behavior. + The toggle integration journeys run with `ENABLE_SMART_ROUTER_ORCHESTRATOR` unset and with `ENABLE_SMART_ROUTER_ORCHESTRATOR=1`. They require only `smart-router` by default and both bundled skills when opted in, verify the saved session controls and native diff --git a/tests/conftest.py b/tests/conftest.py index c68df98e7..c16ab3801 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -74,6 +74,8 @@ def reject_privileged_write(path, _desired_text): # header-rendering tests. Clear them so routing stays off unless a test opts in. monkeypatch.delenv("ENABLE_SMART_ROUTING_V2", raising=False) monkeypatch.delenv("ENABLE_SMART_ROUTING_SUBAGENT_ONLY", raising=False) + monkeypatch.delenv("ENABLE_SMART_ROUTER_ORCHESTRATOR", raising=False) + monkeypatch.delenv("SMART_ROUTING_CONFIG_VERSION", raising=False) monkeypatch.delenv("SMART_ROUTER_NAME", raising=False) # On Windows, resolve_command swaps a bare program name for whatever `shutil.which` # finds on the developer's PATH (e.g. a real `codex.CMD`). Rebind only the compatibility diff --git a/tests/integration/README.md b/tests/integration/README.md index b451cac0e..6e55fc035 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -303,6 +303,11 @@ role-contract preservation, and isolation from legacy preference files lack dedicated regression coverage. Codex's native hook merging, project trust, and execution of pre-existing hooks are not exercised by this integration suite. +The unit/component `../test_smart_routing_config.py` covers selector resolution and legacy +precedence, process-local environment restoration, native-subcommand suppression, and +orchestrator off-to-on transitions through real temporary session files. It does not claim +live agent, hook, or gateway coverage. + The portable `../test_claude_windows_smart_routing.py` checks the Windows subagent-only fallback without Unix imports. Native Windows TUI and hook execution remain outside this integration suite. diff --git a/tests/test_smart_routing_config.py b/tests/test_smart_routing_config.py new file mode 100644 index 000000000..0eff00076 --- /dev/null +++ b/tests/test_smart_routing_config.py @@ -0,0 +1,363 @@ +"""Component coverage for versioned smart-routing configuration.""" + +from __future__ import annotations + +import os +from unittest.mock import patch + +import pytest +import typer +from typer.testing import CliRunner + +import ucode.cli as cli +from ucode.constants import ( + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, + ENABLE_SMART_ROUTING_ENV_VAR, + ENABLE_SUBAGENT_ROUTING_ENV_VAR, + SMART_ROUTING_CONFIG_VERSION_ENV_VAR, +) +from ucode.smart_routing import config, orchestrator, session_env, v2 + +runner = CliRunner() + + +@pytest.mark.parametrize( + ("selector", "expected"), + [ + ( + "subagent_only", + { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + }, + ), + ( + "subagent_orch", + { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + }, + ), + ], +) +def test_resolve_environment_materializes_selector_without_mutating_input(selector, expected): + source = { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + "UNRELATED_SETTING": "preserved", + } + + resolved = config.resolve_environment(source) + + assert resolved == {**expected, "UNRELATED_SETTING": "preserved"} + assert source == { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + "UNRELATED_SETTING": "preserved", + } + + +@pytest.mark.parametrize("selector", [None, "", " \t"]) +def test_resolve_environment_uses_legacy_flags_for_blank_or_unset_selector(selector): + source = { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + } + if selector is not None: + source[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] = selector + original = source.copy() + + resolved = config.resolve_environment(source) + + assert resolved == { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + } + assert source == original + + +def test_legacy_flags_still_drive_routing_queries_without_selector(): + environment = { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + } + + assert config.resolve_environment(environment) == environment + assert v2.smart_routing_enabled(environment) is True + assert v2.first_prompt_routing_enabled(environment) is True + assert orchestrator.feature_enabled(environment) is True + + +@pytest.mark.parametrize("selector", ["subagent_only", "subagent_orch"]) +def test_selector_wins_legacy_conflicts_for_routing_queries(selector): + environment = { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + } + + assert v2.smart_routing_enabled(environment) is True + assert v2.first_prompt_routing_enabled(environment) is False + assert orchestrator.feature_enabled(environment) is (selector == "subagent_orch") + + +def test_apply_config_materializes_selector_and_returns_previous_owned_values(): + environment = { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_only", + ENABLE_SMART_ROUTING_ENV_VAR: "old-v2", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "old-subagent", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "old-orchestrator", + "UNRELATED_SETTING": "preserved", + } + + previous = config.apply_config(environment) + + assert previous == { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_only", + ENABLE_SMART_ROUTING_ENV_VAR: "old-v2", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "old-subagent", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "old-orchestrator", + } + assert environment == { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + "UNRELATED_SETTING": "preserved", + } + + +@pytest.mark.parametrize("selector", [None, "", " "]) +def test_apply_config_is_a_noop_for_unset_or_blank_selector(selector): + environment = {"UNRELATED_SETTING": "preserved"} + if selector is not None: + environment[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] = selector + original = environment.copy() + + assert config.apply_config(environment) == {} + assert environment == original + + +def test_unknown_selector_mentions_supported_names_and_does_not_mutate_input(): + for resolver in (config.resolve_environment, config.apply_config): + environment = {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "future_mode"} + + with pytest.raises(RuntimeError) as caught: + resolver(environment) + + message = str(caught.value) + assert "future_mode" in message + assert "subagent_only" in message + assert "subagent_orch" in message + assert environment == {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "future_mode"} + + +@pytest.mark.parametrize("operation", ["enable", "override", "disable"]) +def test_v2_routing_toggles_restore_selector_and_legacy_environment(operation): + original = { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch", + "UNRELATED_SETTING": "preserved", + } + environment = original.copy() + + if operation == "enable": + previous = v2.enable_smart_routing(environment) + expected = { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + "UNRELATED_SETTING": "preserved", + } + elif operation == "override": + previous = v2.override_smart_routing(True, environment) + expected = { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + "UNRELATED_SETTING": "preserved", + } + else: + previous = v2.disable_smart_routing(environment) + expected = {"UNRELATED_SETTING": "preserved"} + + assert environment == expected + assert SMART_ROUTING_CONFIG_VERSION_ENV_VAR in previous + + v2.restore_smart_routing_env(previous, environment) + + assert environment == original + + +def test_explicit_disable_wins_over_selector_until_restored(): + original = {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch"} + environment = original.copy() + + previous = v2.override_smart_routing(False, environment) + + assert SMART_ROUTING_CONFIG_VERSION_ENV_VAR not in environment + assert environment[ENABLE_SMART_ROUTING_ENV_VAR] == "0" + assert environment[ENABLE_SUBAGENT_ROUTING_ENV_VAR] == "0" + assert v2.smart_routing_enabled(environment) is False + assert v2.first_prompt_routing_enabled(environment) is False + + v2.restore_smart_routing_env(previous, environment) + + assert environment == original + + +def test_cli_context_materializes_inherited_selector_and_restores_after_failure(monkeypatch): + original = { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch", + ENABLE_SMART_ROUTING_ENV_VAR: "old-v2", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "old-subagent", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "old-orchestrator", + } + for key, value in original.items(): + monkeypatch.setenv(key, value) + + with pytest.raises(ValueError, match="launch failed"): + with cli._smart_routing_v2_flag(None): + assert SMART_ROUTING_CONFIG_VERSION_ENV_VAR not in os.environ + assert os.environ[ENABLE_SMART_ROUTING_ENV_VAR] == "0" + assert os.environ[ENABLE_SUBAGENT_ROUTING_ENV_VAR] == "1" + assert os.environ[ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR] == "1" + assert v2.smart_routing_enabled() is True + raise ValueError("launch failed") + + assert {key: os.environ.get(key) for key in original} == original + + +def test_cli_context_turns_invalid_selector_into_actionable_exit(monkeypatch): + monkeypatch.setenv(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "future_mode") + + with pytest.raises(typer.Exit) as caught: + with cli._smart_routing_v2_flag(None): + pytest.fail("invalid selector should prevent entering the context") + + assert caught.value.exit_code == 1 + assert os.environ[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] == "future_mode" + + +def test_bare_launch_materializes_selector_for_managed_default_and_restores_environment( + monkeypatch, +): + original = { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch", + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + } + for key, value in original.items(): + monkeypatch.setenv(key, value) + observed = [] + + def inspect_managed_default(*_args, **_kwargs): + observed.append( + { + "selector": os.environ.get(SMART_ROUTING_CONFIG_VERSION_ENV_VAR), + "v2": os.environ.get(ENABLE_SMART_ROUTING_ENV_VAR), + "subagent": os.environ.get(ENABLE_SUBAGENT_ROUTING_ENV_VAR), + "orchestrator": os.environ.get(ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR), + } + ) + + with patch.object( + cli, "_launch_managed_default", side_effect=inspect_managed_default + ) as launch: + result = runner.invoke(cli.app, []) + + assert result.exit_code == 0, result.output + launch.assert_called_once() + assert observed == [ + { + "selector": None, + "v2": "0", + "subagent": "1", + "orchestrator": "1", + } + ] + assert {key: os.environ.get(key) for key in original} == original + + +def test_bare_launch_rejects_invalid_selector_before_managed_default(monkeypatch): + monkeypatch.setenv(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "future_mode") + + with patch.object(cli, "_launch_managed_default") as launch: + result = runner.invoke(cli.app, []) + + assert result.exit_code == 1 + launch.assert_not_called() + assert "subagent_only" in result.output + assert "subagent_orch" in result.output + + +@pytest.mark.parametrize( + ("tool", "subcommand"), + [("codex", "app"), ("claude", "update")], +) +def test_native_subcommand_suppresses_inherited_selector_routing(monkeypatch, tool, subcommand): + monkeypatch.setenv(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "subagent_orch") + observed = [] + + with patch( + "ucode.cli._launch_tool", + side_effect=lambda *_args, **_kwargs: observed.append( + { + "selector": os.environ.get(SMART_ROUTING_CONFIG_VERSION_ENV_VAR), + "v2": os.environ.get(ENABLE_SMART_ROUTING_ENV_VAR), + "subagent": os.environ.get(ENABLE_SUBAGENT_ROUTING_ENV_VAR), + "enabled": v2.smart_routing_enabled(), + } + ), + ): + result = runner.invoke(cli.app, [tool, subcommand]) + + assert result.exit_code == 0, result.output + assert observed == [{"selector": None, "v2": None, "subagent": None, "enabled": False}] + assert os.environ[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] == "subagent_orch" + + +def test_session_file_overrides_resolved_selector_for_off_to_on_orchestration( + tmp_path, monkeypatch +): + session_path = tmp_path / "session-env.json" + session_path.write_text("{}", encoding="utf-8") + environment = { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch", + session_env.SESSION_ENV_VAR: str(session_path), + } + monkeypatch.setenv(session_env.SESSION_ENV_VAR, str(session_path)) + + session_env.set_session_environment( + { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + } + ) + off = session_env.effective_environment(environment) + + assert off[ENABLE_SMART_ROUTING_ENV_VAR] == "0" + assert off[ENABLE_SUBAGENT_ROUTING_ENV_VAR] == "0" + assert off[ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR] == "1" + assert v2.smart_routing_enabled(off) is False + assert orchestrator.enabled(environment) is False + + session_env.set_session_environment({}) + on = session_env.effective_environment(environment) + + assert on[ENABLE_SMART_ROUTING_ENV_VAR] == "0" + assert on[ENABLE_SUBAGENT_ROUTING_ENV_VAR] == "1" + assert on[ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR] == "1" + assert v2.smart_routing_enabled(on) is True + assert v2.first_prompt_routing_enabled(on) is False + assert orchestrator.enabled(environment) is True + assert environment[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] == "subagent_orch" From 3913637ad24a7034a6ce41355f8c0bc2515a3988 Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Thu, 8 Oct 2026 18:35:18 +0000 Subject: [PATCH 02/30] Version smart routing presets and validate complete flag definitions --- README.md | 17 +++- skills/smart-router-orchestrator/README.md | 5 +- src/ucode/constants.py | 4 + src/ucode/smart_routing/config.py | 36 +++++++- tests/README.md | 9 +- tests/integration/README.md | 8 +- tests/test_smart_routing_config.py | 99 +++++++++++++++++++--- 7 files changed, 150 insertions(+), 28 deletions(-) diff --git a/README.md b/README.md index aeb923d89..31561c2c7 100644 --- a/README.md +++ b/README.md @@ -252,15 +252,15 @@ Use `SMART_ROUTING_CONFIG_VERSION` at launch to select a smart-routing configura | Version | Subagent routing | First-prompt routing | Orchestrator | | --- | --- | --- | --- | -| `subagent_only` | On | Off | Off | -| `subagent_orch` | On | Off | On | +| `subagent_only_v0` | On | Off | Off | +| `subagent_orch_v0` | On | Off | On | -Smart-routed Claude and Codex sessions install `smart-router`. The `subagent_orch` +Smart-routed Claude and Codex sessions install `smart-router`. The `subagent_orch_v0` version also installs and activates the bundled `smart-router-orchestrator` skill. For example: ```bash -SMART_ROUTING_CONFIG_VERSION=subagent_orch ug claude +SMART_ROUTING_CONFIG_VERSION=subagent_orch_v0 ug claude ``` The version takes precedence over conflicting legacy flags. UG expands it into @@ -270,6 +270,15 @@ unset or empty, these legacy flags retain their existing behavior, including first-prompt routing through `ENABLE_SMART_ROUTING_V2=1`. Unknown versions produce an error listing the supported values. Orchestration remains off by default. +The original `subagent_only` and `subagent_orch` names remain aliases of their +respective `_v0` configurations. Future revisions use new `_v1`, `_v2`, etc. names +without changing existing versions or aliases. + +Version definitions fail validation at module import if any flag in +`SMART_ROUTING_CONFIG_ENV_KEYS` is missing, has a value other than `"0"` or `"1"`, +or an unknown flag is present. Register new managed flags in that tuple and +explicitly set them in every version. + Use `ug codex` in the same command for Codex. Smart Router Orchestrator assigns bounded work to explorer, researcher, worker, tester, and reviewer roles while the root plans, integrates, and verifies results. Easy tasks and explicit requests not to delegate diff --git a/skills/smart-router-orchestrator/README.md b/skills/smart-router-orchestrator/README.md index 2fb507f91..fb5f618a6 100644 --- a/skills/smart-router-orchestrator/README.md +++ b/skills/smart-router-orchestrator/README.md @@ -2,10 +2,11 @@ UG bundles the `smart-router-orchestrator` workflow and five Claude role definitions. Smart-routed Claude and Codex launches install and activate this skill alongside -`smart-router` with `SMART_ROUTING_CONFIG_VERSION=subagent_orch`. +`smart-router` with `SMART_ROUTING_CONFIG_VERSION=subagent_orch_v0`. UG expands that version into the session's legacy feature flags. The existing `ENABLE_SMART_ROUTER_ORCHESTRATOR=1` opt-in remains supported when -`SMART_ROUTING_CONFIG_VERSION` is unset; `subagent_only` explicitly leaves orchestration off. +`SMART_ROUTING_CONFIG_VERSION` is unset; `subagent_only_v0` explicitly leaves orchestration off. +The original `subagent_orch` and `subagent_only` names remain aliases of these `_v0` versions. The feature is off by default; routing alone installs only `smart-router`. The workflow is injected before root prompts and after compaction. The hook checks diff --git a/src/ucode/constants.py b/src/ucode/constants.py index 590f94f33..fbbbdc6a1 100644 --- a/src/ucode/constants.py +++ b/src/ucode/constants.py @@ -11,6 +11,10 @@ ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, ) +SMART_ROUTING_CONFIG_ENV_KEYS = ( + *SMART_ROUTING_ENV_KEYS, + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, +) MODEL_PROVIDER_SERVICE_HEADER = "Databricks-Model-Provider-Service" MODEL_SERVICE_PARENT_SCHEMA_HEADER = "Databricks-Model-Service-Parent-Schema" diff --git a/src/ucode/smart_routing/config.py b/src/ucode/smart_routing/config.py index 8c71f6fa3..0fa0e517b 100644 --- a/src/ucode/smart_routing/config.py +++ b/src/ucode/smart_routing/config.py @@ -9,21 +9,48 @@ ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, + SMART_ROUTING_CONFIG_ENV_KEYS, SMART_ROUTING_CONFIG_VERSION_ENV_VAR, ) _VERSIONS = { - "subagent_only": { + "subagent_only_v0": { ENABLE_SMART_ROUTING_ENV_VAR: "0", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", }, - "subagent_orch": { + "subagent_orch_v0": { ENABLE_SMART_ROUTING_ENV_VAR: "0", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", }, } +_VERSION_ALIASES = { + "subagent_only": "subagent_only_v0", + "subagent_orch": "subagent_orch_v0", +} + + +def _validate_versions(versions: Mapping[str, Mapping[str, str]]) -> None: + """Require every version to explicitly configure the complete managed flag set.""" + expected = set(SMART_ROUTING_CONFIG_ENV_KEYS) + for version, values in versions.items(): + missing = expected - values.keys() + unexpected = values.keys() - expected + if missing or unexpected: + raise ValueError( + f"Invalid smart-routing version {version!r}: " + f"missing env vars {sorted(missing)}; unexpected env vars {sorted(unexpected)}." + ) + for key, value in values.items(): + if value not in ("0", "1"): + raise ValueError( + f"Invalid smart-routing version {version!r}: " + f"{key} must be '0' or '1', got {value!r}." + ) + + +_validate_versions(_VERSIONS) def resolve_environment(env: Mapping[str, str] | None = None) -> dict[str, str]: @@ -32,6 +59,7 @@ def resolve_environment(env: Mapping[str, str] | None = None) -> dict[str, str]: version = resolved.pop(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "").strip() if not version: return resolved + version = _VERSION_ALIASES.get(version, version) if version not in _VERSIONS: raise RuntimeError( f"Unknown {SMART_ROUTING_CONFIG_VERSION_ENV_VAR} value {version!r}. " @@ -48,8 +76,8 @@ def apply_config(env: MutableMapping[str, str] | None = None) -> dict[str, str | if not version: return {} resolved = resolve_environment(target) - keys = (*_VERSIONS[version], SMART_ROUTING_CONFIG_VERSION_ENV_VAR) + keys = (*SMART_ROUTING_CONFIG_ENV_KEYS, SMART_ROUTING_CONFIG_VERSION_ENV_VAR) previous = {key: target.get(key) for key in keys} - target.update({key: resolved[key] for key in _VERSIONS[version]}) + target.update({key: resolved[key] for key in SMART_ROUTING_CONFIG_ENV_KEYS}) target.pop(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, None) return previous diff --git a/tests/README.md b/tests/README.md index bb7500cfd..9203582a7 100644 --- a/tests/README.md +++ b/tests/README.md @@ -135,10 +135,11 @@ that Claude settings and Codex's shell policy carry the interpreter and session These are component checks; they do not establish native skill permission matching or PowerShell execution. -`test_smart_routing_config.py` covers versioned selector resolution, legacy-flag precedence, -environment materialization/restoration, native-subcommand suppression, and orchestrator -state transitions through real temporary session files. These are component checks; they do -not establish live agent, hook, or gateway behavior. +`test_smart_routing_config.py` covers canonical v0 and legacy selector resolution, legacy-flag +precedence, fail-fast registry validation, environment materialization/restoration, +native-subcommand suppression, and orchestrator state transitions through real temporary +session files. These are component checks; they do not establish live agent, hook, or gateway +behavior. The toggle integration journeys run with `ENABLE_SMART_ROUTER_ORCHESTRATOR` unset and with `ENABLE_SMART_ROUTER_ORCHESTRATOR=1`. They require only `smart-router` by default and both diff --git a/tests/integration/README.md b/tests/integration/README.md index 6e55fc035..d5b6df42b 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -303,10 +303,10 @@ role-contract preservation, and isolation from legacy preference files lack dedicated regression coverage. Codex's native hook merging, project trust, and execution of pre-existing hooks are not exercised by this integration suite. -The unit/component `../test_smart_routing_config.py` covers selector resolution and legacy -precedence, process-local environment restoration, native-subcommand suppression, and -orchestrator off-to-on transitions through real temporary session files. It does not claim -live agent, hook, or gateway coverage. +The unit/component `../test_smart_routing_config.py` covers canonical-v0 and legacy selector +resolution, registry validation, process-local environment restoration, +native-subcommand suppression, and orchestrator off-to-on transitions through real temporary +session files. It does not claim live agent, hook, or gateway coverage. The portable `../test_claude_windows_smart_routing.py` checks the Windows subagent-only fallback without Unix imports. Native Windows TUI and hook execution diff --git a/tests/test_smart_routing_config.py b/tests/test_smart_routing_config.py index 0eff00076..934989bfb 100644 --- a/tests/test_smart_routing_config.py +++ b/tests/test_smart_routing_config.py @@ -3,6 +3,7 @@ from __future__ import annotations import os +import runpy from unittest.mock import patch import pytest @@ -10,10 +11,12 @@ from typer.testing import CliRunner import ucode.cli as cli +import ucode.constants as constants from ucode.constants import ( ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, + SMART_ROUTING_CONFIG_ENV_KEYS, SMART_ROUTING_CONFIG_VERSION_ENV_VAR, ) from ucode.smart_routing import config, orchestrator, session_env, v2 @@ -63,6 +66,34 @@ def test_resolve_environment_materializes_selector_without_mutating_input(select } +@pytest.mark.parametrize( + ("selector", "expected_orchestrator"), + [("subagent_only_v0", "0"), ("subagent_orch_v0", "1")], +) +def test_resolve_environment_supports_canonical_v0_selectors(selector, expected_orchestrator): + resolved = config.resolve_environment({SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector}) + + assert resolved == { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: expected_orchestrator, + } + + +@pytest.mark.parametrize( + ("alias", "canonical"), + [ + ("subagent_only", "subagent_only_v0"), + ("subagent_orch", "subagent_orch_v0"), + ], +) +def test_unsuffixed_selectors_alias_fixed_v0_definitions(alias, canonical): + assert config._VERSION_ALIASES[alias] == canonical + assert config.resolve_environment( + {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: alias} + ) == config.resolve_environment({SMART_ROUTING_CONFIG_VERSION_ENV_VAR: canonical}) + + @pytest.mark.parametrize("selector", [None, "", " \t"]) def test_resolve_environment_uses_legacy_flags_for_blank_or_unset_selector(selector): source = { @@ -113,7 +144,7 @@ def test_selector_wins_legacy_conflicts_for_routing_queries(selector): def test_apply_config_materializes_selector_and_returns_previous_owned_values(): environment = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_only", + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_only_v0", ENABLE_SMART_ROUTING_ENV_VAR: "old-v2", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "old-subagent", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "old-orchestrator", @@ -122,8 +153,12 @@ def test_apply_config_materializes_selector_and_returns_previous_owned_values(): previous = config.apply_config(environment) + assert set(previous) == { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR, + *SMART_ROUTING_CONFIG_ENV_KEYS, + } assert previous == { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_only", + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_only_v0", ENABLE_SMART_ROUTING_ENV_VAR: "old-v2", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "old-subagent", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "old-orchestrator", @@ -136,6 +171,13 @@ def test_apply_config_materializes_selector_and_returns_previous_owned_values(): } +@pytest.mark.parametrize("selector", ["subagent_only_v1", "subagent_orch_v1"]) +@pytest.mark.parametrize("resolver", [config.resolve_environment, config.apply_config]) +def test_future_v1_selectors_remain_unknown(selector, resolver): + with pytest.raises(RuntimeError): + resolver({SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector}) + + @pytest.mark.parametrize("selector", [None, "", " "]) def test_apply_config_is_a_noop_for_unset_or_blank_selector(selector): environment = {"UNRELATED_SETTING": "preserved"} @@ -161,10 +203,47 @@ def test_unknown_selector_mentions_supported_names_and_does_not_mutate_input(): assert environment == {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "future_mode"} +@pytest.mark.parametrize("missing_key", SMART_ROUTING_CONFIG_ENV_KEYS) +def test_validate_versions_rejects_each_missing_managed_flag(missing_key): + values = {key: "0" for key in SMART_ROUTING_CONFIG_ENV_KEYS if key != missing_key} + + with pytest.raises(ValueError): + config._validate_versions({"test_version": values}) + + +@pytest.mark.parametrize("invalid_value", [None, "", "2", "true", 0, False]) +def test_validate_versions_rejects_non_binary_managed_flag_values(invalid_value): + values = dict.fromkeys(SMART_ROUTING_CONFIG_ENV_KEYS, "0") + values[ENABLE_SMART_ROUTING_ENV_VAR] = invalid_value + + with pytest.raises(ValueError): + config._validate_versions({"test_version": values}) + + +def test_validate_versions_rejects_unexpected_managed_flag(): + values = dict.fromkeys(SMART_ROUTING_CONFIG_ENV_KEYS, "0") + values["UNEXPECTED_SMART_ROUTING_FLAG"] = "0" + + with pytest.raises(ValueError): + config._validate_versions({"test_version": values}) + + +def test_config_import_rejects_new_registry_flag_before_runtime_use(monkeypatch): + new_key = "ENABLE_SMART_ROUTING_TEST_ONLY" + monkeypatch.setattr( + constants, + "SMART_ROUTING_CONFIG_ENV_KEYS", + (*constants.SMART_ROUTING_CONFIG_ENV_KEYS, new_key), + ) + + with pytest.raises(ValueError): + runpy.run_path(config.__file__) + + @pytest.mark.parametrize("operation", ["enable", "override", "disable"]) def test_v2_routing_toggles_restore_selector_and_legacy_environment(operation): original = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch", + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch_v0", "UNRELATED_SETTING": "preserved", } environment = original.copy() @@ -198,7 +277,7 @@ def test_v2_routing_toggles_restore_selector_and_legacy_environment(operation): def test_explicit_disable_wins_over_selector_until_restored(): - original = {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch"} + original = {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch_v0"} environment = original.copy() previous = v2.override_smart_routing(False, environment) @@ -216,7 +295,7 @@ def test_explicit_disable_wins_over_selector_until_restored(): def test_cli_context_materializes_inherited_selector_and_restores_after_failure(monkeypatch): original = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch", + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch_v0", ENABLE_SMART_ROUTING_ENV_VAR: "old-v2", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "old-subagent", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "old-orchestrator", @@ -251,7 +330,7 @@ def test_bare_launch_materializes_selector_for_managed_default_and_restores_envi monkeypatch, ): original = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch", + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch_v0", ENABLE_SMART_ROUTING_ENV_VAR: "1", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", @@ -305,7 +384,7 @@ def test_bare_launch_rejects_invalid_selector_before_managed_default(monkeypatch [("codex", "app"), ("claude", "update")], ) def test_native_subcommand_suppresses_inherited_selector_routing(monkeypatch, tool, subcommand): - monkeypatch.setenv(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "subagent_orch") + monkeypatch.setenv(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "subagent_orch_v0") observed = [] with patch( @@ -323,7 +402,7 @@ def test_native_subcommand_suppresses_inherited_selector_routing(monkeypatch, to assert result.exit_code == 0, result.output assert observed == [{"selector": None, "v2": None, "subagent": None, "enabled": False}] - assert os.environ[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] == "subagent_orch" + assert os.environ[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] == "subagent_orch_v0" def test_session_file_overrides_resolved_selector_for_off_to_on_orchestration( @@ -332,7 +411,7 @@ def test_session_file_overrides_resolved_selector_for_off_to_on_orchestration( session_path = tmp_path / "session-env.json" session_path.write_text("{}", encoding="utf-8") environment = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch", + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch_v0", session_env.SESSION_ENV_VAR: str(session_path), } monkeypatch.setenv(session_env.SESSION_ENV_VAR, str(session_path)) @@ -360,4 +439,4 @@ def test_session_file_overrides_resolved_selector_for_off_to_on_orchestration( assert v2.smart_routing_enabled(on) is True assert v2.first_prompt_routing_enabled(on) is False assert orchestrator.enabled(environment) is True - assert environment[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] == "subagent_orch" + assert environment[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] == "subagent_orch_v0" From 9bbd850c01680a85a6f829ecf2fbe5d37fe67c11 Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Thu, 8 Oct 2026 20:04:19 +0000 Subject: [PATCH 03/30] Require explicit smart routing version names without aliases --- README.md | 5 ++- skills/smart-router-orchestrator/README.md | 1 - src/ucode/smart_routing/config.py | 5 --- tests/README.md | 4 +-- tests/integration/README.md | 4 +-- tests/test_smart_routing_config.py | 41 +++++++++++----------- 6 files changed, 27 insertions(+), 33 deletions(-) diff --git a/README.md b/README.md index 31561c2c7..12230e4c2 100644 --- a/README.md +++ b/README.md @@ -270,9 +270,8 @@ unset or empty, these legacy flags retain their existing behavior, including first-prompt routing through `ENABLE_SMART_ROUTING_V2=1`. Unknown versions produce an error listing the supported values. Orchestration remains off by default. -The original `subagent_only` and `subagent_orch` names remain aliases of their -respective `_v0` configurations. Future revisions use new `_v1`, `_v2`, etc. names -without changing existing versions or aliases. +Version names require an explicit suffix. Future revisions use new `_v1`, `_v2`, +etc. names without changing existing versions. Version definitions fail validation at module import if any flag in `SMART_ROUTING_CONFIG_ENV_KEYS` is missing, has a value other than `"0"` or `"1"`, diff --git a/skills/smart-router-orchestrator/README.md b/skills/smart-router-orchestrator/README.md index fb5f618a6..32d30bdd8 100644 --- a/skills/smart-router-orchestrator/README.md +++ b/skills/smart-router-orchestrator/README.md @@ -6,7 +6,6 @@ Smart-routed Claude and Codex launches install and activate this skill alongside UG expands that version into the session's legacy feature flags. The existing `ENABLE_SMART_ROUTER_ORCHESTRATOR=1` opt-in remains supported when `SMART_ROUTING_CONFIG_VERSION` is unset; `subagent_only_v0` explicitly leaves orchestration off. -The original `subagent_orch` and `subagent_only` names remain aliases of these `_v0` versions. The feature is off by default; routing alone installs only `smart-router`. The workflow is injected before root prompts and after compaction. The hook checks diff --git a/src/ucode/smart_routing/config.py b/src/ucode/smart_routing/config.py index 0fa0e517b..b8cb26bb8 100644 --- a/src/ucode/smart_routing/config.py +++ b/src/ucode/smart_routing/config.py @@ -25,10 +25,6 @@ ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", }, } -_VERSION_ALIASES = { - "subagent_only": "subagent_only_v0", - "subagent_orch": "subagent_orch_v0", -} def _validate_versions(versions: Mapping[str, Mapping[str, str]]) -> None: @@ -59,7 +55,6 @@ def resolve_environment(env: Mapping[str, str] | None = None) -> dict[str, str]: version = resolved.pop(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "").strip() if not version: return resolved - version = _VERSION_ALIASES.get(version, version) if version not in _VERSIONS: raise RuntimeError( f"Unknown {SMART_ROUTING_CONFIG_VERSION_ENV_VAR} value {version!r}. " diff --git a/tests/README.md b/tests/README.md index 9203582a7..c217f8798 100644 --- a/tests/README.md +++ b/tests/README.md @@ -135,8 +135,8 @@ that Claude settings and Codex's shell policy carry the interpreter and session These are component checks; they do not establish native skill permission matching or PowerShell execution. -`test_smart_routing_config.py` covers canonical v0 and legacy selector resolution, legacy-flag -precedence, fail-fast registry validation, environment materialization/restoration, +`test_smart_routing_config.py` covers canonical v0 selector resolution, legacy-flag precedence, +fail-fast registry validation, environment materialization/restoration, native-subcommand suppression, and orchestrator state transitions through real temporary session files. These are component checks; they do not establish live agent, hook, or gateway behavior. diff --git a/tests/integration/README.md b/tests/integration/README.md index d5b6df42b..1c07acfa6 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -303,8 +303,8 @@ role-contract preservation, and isolation from legacy preference files lack dedicated regression coverage. Codex's native hook merging, project trust, and execution of pre-existing hooks are not exercised by this integration suite. -The unit/component `../test_smart_routing_config.py` covers canonical-v0 and legacy selector -resolution, registry validation, process-local environment restoration, +The unit/component `../test_smart_routing_config.py` covers canonical-v0 selector resolution, +legacy-flag precedence, registry validation, process-local environment restoration, native-subcommand suppression, and orchestrator off-to-on transitions through real temporary session files. It does not claim live agent, hook, or gateway coverage. diff --git a/tests/test_smart_routing_config.py b/tests/test_smart_routing_config.py index 934989bfb..57aea1809 100644 --- a/tests/test_smart_routing_config.py +++ b/tests/test_smart_routing_config.py @@ -28,7 +28,7 @@ ("selector", "expected"), [ ( - "subagent_only", + "subagent_only_v0", { ENABLE_SMART_ROUTING_ENV_VAR: "0", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", @@ -36,7 +36,7 @@ }, ), ( - "subagent_orch", + "subagent_orch_v0", { ENABLE_SMART_ROUTING_ENV_VAR: "0", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", @@ -80,18 +80,19 @@ def test_resolve_environment_supports_canonical_v0_selectors(selector, expected_ } -@pytest.mark.parametrize( - ("alias", "canonical"), - [ - ("subagent_only", "subagent_only_v0"), - ("subagent_orch", "subagent_orch_v0"), - ], -) -def test_unsuffixed_selectors_alias_fixed_v0_definitions(alias, canonical): - assert config._VERSION_ALIASES[alias] == canonical - assert config.resolve_environment( - {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: alias} - ) == config.resolve_environment({SMART_ROUTING_CONFIG_VERSION_ENV_VAR: canonical}) +@pytest.mark.parametrize("selector", ["subagent_only", "subagent_orch"]) +@pytest.mark.parametrize("resolver", [config.resolve_environment, config.apply_config]) +def test_unsuffixed_selectors_are_rejected_without_mutating_input(selector, resolver): + environment = { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, + "UNRELATED_SETTING": "preserved", + } + original = environment.copy() + + with pytest.raises(RuntimeError): + resolver(environment) + + assert environment == original @pytest.mark.parametrize("selector", [None, "", " \t"]) @@ -128,7 +129,7 @@ def test_legacy_flags_still_drive_routing_queries_without_selector(): assert orchestrator.feature_enabled(environment) is True -@pytest.mark.parametrize("selector", ["subagent_only", "subagent_orch"]) +@pytest.mark.parametrize("selector", ["subagent_only_v0", "subagent_orch_v0"]) def test_selector_wins_legacy_conflicts_for_routing_queries(selector): environment = { SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, @@ -139,7 +140,7 @@ def test_selector_wins_legacy_conflicts_for_routing_queries(selector): assert v2.smart_routing_enabled(environment) is True assert v2.first_prompt_routing_enabled(environment) is False - assert orchestrator.feature_enabled(environment) is (selector == "subagent_orch") + assert orchestrator.feature_enabled(environment) is (selector == "subagent_orch_v0") def test_apply_config_materializes_selector_and_returns_previous_owned_values(): @@ -198,8 +199,8 @@ def test_unknown_selector_mentions_supported_names_and_does_not_mutate_input(): message = str(caught.value) assert "future_mode" in message - assert "subagent_only" in message - assert "subagent_orch" in message + assert "subagent_only_v0" in message + assert "subagent_orch_v0" in message assert environment == {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "future_mode"} @@ -375,8 +376,8 @@ def test_bare_launch_rejects_invalid_selector_before_managed_default(monkeypatch assert result.exit_code == 1 launch.assert_not_called() - assert "subagent_only" in result.output - assert "subagent_orch" in result.output + assert "subagent_only_v0" in result.output + assert "subagent_orch_v0" in result.output @pytest.mark.parametrize( From 34c00c85b35c4e0e7a8bb8eca8e094ea9b0fa054 Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Thu, 8 Oct 2026 20:07:47 +0000 Subject: [PATCH 04/30] Document how to extend smart routing configurations --- AGENTS.md | 41 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 41 insertions(+) diff --git a/AGENTS.md b/AGENTS.md index 156609e5f..ea8c30848 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -29,6 +29,47 @@ Tests live in `tests/`. - Add or update focused tests for behavior changes. - Do not modify generated or lock files unless the dependency graph intentionally changes. +## Smart-routing configuration + +`SMART_ROUTING_CONFIG_VERSION` is the external selector. Version definitions live in +`src/ucode/smart_routing/config.py` under `_VERSIONS`; their managed environment keys +are registered in `SMART_ROUTING_CONFIG_ENV_KEYS` in `src/ucode/constants.py`. +An unset or empty selector preserves legacy environment-flag behavior. + +### Adding a parameter + +1. Define its environment-variable constant in `src/ucode/constants.py` and add it to + `SMART_ROUTING_CONFIG_ENV_KEYS`. +2. Set an explicit value for it in **every** `_VERSIONS` entry, including existing versions. + Choose values that preserve existing versions' behavior. Current parameters accept only + the strings `"0"` and `"1"`; do not use booleans, empty strings, or omitted keys. +3. Add its consumer in the appropriate routing module. Use `resolve_environment` for + config-aware reads, or the legacy flags materialized by `apply_config` at launch. + Keep launch-scoped changes restorable and preserve legacy behavior without a selector. +4. Do not add arbitrary parameters to `SMART_ROUTING_ENV_KEYS`: that separate tuple controls + routing activation and session on/off overrides. Extend it only when those semantics + are intended, with regression tests. Session overrides must apply after version resolution. + +`_validate_versions` runs at module import and rejects missing keys, unknown keys, and +invalid values. Do not weaken the complete-key check or infer required keys from `_VERSIONS`. +If a new parameter needs nonbinary values, add parameter-specific validation and tests. + +### Adding a config type or revision + +1. Add a complete mapping to `_VERSIONS` with an explicit suffix, such as `new_mode_v0`. + For a changed existing mode, add `existing_mode_v1` rather than changing its `_v0` behavior. + Do not add unsuffixed names or version aliases. +2. Define every key in `SMART_ROUTING_CONFIG_ENV_KEYS`; never rely on the caller's + inherited environment to fill missing values. +3. Update the version table and examples in `README.md` and any affected bundled-skill docs. +4. Extend `tests/test_smart_routing_config.py` for the new mode, precedence over legacy flags, + environment restoration, validation failures, and applicable session/launch behavior. + Update unknown-version tests when a previously rejected version becomes supported. + Follow `tests/AGENTS.md` and update its coverage READMEs. + +Run `uv run pytest tests/test_smart_routing_config.py` plus relevant routing/CLI tests, +then `just lint`. These component checks do not establish live agent or gateway coverage. + ## Style - Keep user-facing CLI errors actionable. From e8377eaeaf4aa3027105550f58b14f2e8e9cc72f Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Thu, 8 Oct 2026 20:16:31 +0000 Subject: [PATCH 05/30] Verify smart routing versions override all legacy flags --- AGENTS.md | 3 + tests/README.md | 5 +- tests/integration/README.md | 6 +- tests/test_smart_routing_config.py | 90 ++++++++++++++++++++++++++++++ 4 files changed, 100 insertions(+), 4 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index ea8c30848..5944453ce 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -35,6 +35,9 @@ Tests live in `tests/`. `src/ucode/smart_routing/config.py` under `_VERSIONS`; their managed environment keys are registered in `SMART_ROUTING_CONFIG_ENV_KEYS` in `src/ucode/constants.py`. An unset or empty selector preserves legacy environment-flag behavior. +A valid nonempty selector overrides every conflicting legacy value in that registry. +Do not use `setdefault` or preserve inherited values for version-owned parameters. +Explicit launch/session on/off controls still apply after version expansion. ### Adding a parameter diff --git a/tests/README.md b/tests/README.md index c217f8798..366270ef4 100644 --- a/tests/README.md +++ b/tests/README.md @@ -135,8 +135,9 @@ that Claude settings and Codex's shell policy carry the interpreter and session These are component checks; they do not establish native skill permission matching or PowerShell execution. -`test_smart_routing_config.py` covers canonical v0 selector resolution, legacy-flag precedence, -fail-fast registry validation, environment materialization/restoration, +`test_smart_routing_config.py` covers both canonical v0 selectors against all eight binary +legacy-flag combinations, exact precedence/materialization/restoration and getter results, +CLI launch-context materialization/restoration, fail-fast registry validation, native-subcommand suppression, and orchestrator state transitions through real temporary session files. These are component checks; they do not establish live agent, hook, or gateway behavior. diff --git a/tests/integration/README.md b/tests/integration/README.md index 1c07acfa6..71fa8becd 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -303,8 +303,10 @@ role-contract preservation, and isolation from legacy preference files lack dedicated regression coverage. Codex's native hook merging, project trust, and execution of pre-existing hooks are not exercised by this integration suite. -The unit/component `../test_smart_routing_config.py` covers canonical-v0 selector resolution, -legacy-flag precedence, registry validation, process-local environment restoration, +The unit/component `../test_smart_routing_config.py` covers both canonical-v0 selectors against +all eight binary legacy-flag combinations, exact precedence/materialization/restoration and +getter results, CLI launch-context materialization/restoration, registry validation, +process-local environment restoration, native-subcommand suppression, and orchestrator off-to-on transitions through real temporary session files. It does not claim live agent, hook, or gateway coverage. diff --git a/tests/test_smart_routing_config.py b/tests/test_smart_routing_config.py index 57aea1809..96a874bf3 100644 --- a/tests/test_smart_routing_config.py +++ b/tests/test_smart_routing_config.py @@ -4,6 +4,7 @@ import os import runpy +from itertools import product from unittest.mock import patch import pytest @@ -23,6 +24,27 @@ runner = CliRunner() +_SELECTOR_EXPECTED_FLAGS = { + "subagent_only_v0": { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + }, + "subagent_orch_v0": { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + }, +} + + +def _assert_routing_getters(environment: dict[str, str], expected_flags: dict[str, str]) -> None: + assert v2.smart_routing_enabled(environment) is True + assert v2.first_prompt_routing_enabled(environment) is False + assert orchestrator.feature_enabled(environment) is ( + expected_flags[ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR] == "1" + ) + @pytest.mark.parametrize( ("selector", "expected"), @@ -66,6 +88,49 @@ def test_resolve_environment_materializes_selector_without_mutating_input(select } +@pytest.mark.parametrize( + ("selector", "legacy_values", "expected_flags"), + [ + (selector, legacy_values, expected_flags) + for selector, expected_flags in _SELECTOR_EXPECTED_FLAGS.items() + for legacy_values in product(("0", "1"), repeat=len(SMART_ROUTING_CONFIG_ENV_KEYS)) + ], +) +def test_selector_precedence_covers_every_legacy_flag_combination( + selector, legacy_values, expected_flags +): + legacy_environment = dict(zip(SMART_ROUTING_CONFIG_ENV_KEYS, legacy_values, strict=True)) + original = { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, + **legacy_environment, + } + + _assert_routing_getters(original, expected_flags) + + resolved = config.resolve_environment(original) + + assert resolved == expected_flags + _assert_routing_getters(resolved, expected_flags) + assert original == { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, + **legacy_environment, + } + + applied = original.copy() + previous = config.apply_config(applied) + + assert applied == expected_flags + assert previous == { + **legacy_environment, + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, + } + _assert_routing_getters(applied, expected_flags) + + v2.restore_smart_routing_env(previous, applied) + + assert applied == original + + @pytest.mark.parametrize( ("selector", "expected_orchestrator"), [("subagent_only_v0", "0"), ("subagent_orch_v0", "1")], @@ -316,6 +381,31 @@ def test_cli_context_materializes_inherited_selector_and_restores_after_failure( assert {key: os.environ.get(key) for key in original} == original +@pytest.mark.parametrize( + ("selector", "expected_flags"), + list(_SELECTOR_EXPECTED_FLAGS.items()), +) +def test_cli_context_materializes_each_selector_over_opposite_legacy_flags( + monkeypatch, selector, expected_flags +): + legacy_environment = { + key: "1" if value == "0" else "0" for key, value in expected_flags.items() + } + original = { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, + **legacy_environment, + } + for key, value in original.items(): + monkeypatch.setenv(key, value) + + with cli._smart_routing_v2_flag(None): + assert SMART_ROUTING_CONFIG_VERSION_ENV_VAR not in os.environ + assert {key: os.environ.get(key) for key in SMART_ROUTING_CONFIG_ENV_KEYS} == expected_flags + _assert_routing_getters(dict(os.environ), expected_flags) + + assert {key: os.environ.get(key) for key in original} == original + + def test_cli_context_turns_invalid_selector_into_actionable_exit(monkeypatch): monkeypatch.setenv(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "future_mode") From 6614bd1c9e5b2290282097a06aa3131d71a377cf Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Thu, 8 Oct 2026 20:25:52 +0000 Subject: [PATCH 06/30] Include orchestrator in smart routing environment controls --- AGENTS.md | 9 +++++---- src/ucode/constants.py | 4 +--- src/ucode/smart_routing/v2.py | 4 +++- tests/README.md | 3 ++- tests/integration/README.md | 3 ++- tests/test_smart_router.py | 10 +++++++++- tests/test_smart_routing_config.py | 25 +++++++++++++++++++++++++ 7 files changed, 47 insertions(+), 11 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 5944453ce..c2c558fd0 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -42,16 +42,17 @@ Explicit launch/session on/off controls still apply after version expansion. ### Adding a parameter 1. Define its environment-variable constant in `src/ucode/constants.py` and add it to - `SMART_ROUTING_CONFIG_ENV_KEYS`. + `SMART_ROUTING_ENV_KEYS`, which also defines `SMART_ROUTING_CONFIG_ENV_KEYS`. 2. Set an explicit value for it in **every** `_VERSIONS` entry, including existing versions. Choose values that preserve existing versions' behavior. Current parameters accept only the strings `"0"` and `"1"`; do not use booleans, empty strings, or omitted keys. 3. Add its consumer in the appropriate routing module. Use `resolve_environment` for config-aware reads, or the legacy flags materialized by `apply_config` at launch. Keep launch-scoped changes restorable and preserve legacy behavior without a selector. -4. Do not add arbitrary parameters to `SMART_ROUTING_ENV_KEYS`: that separate tuple controls - routing activation and session on/off overrides. Extend it only when those semantics - are intended, with regression tests. Session overrides must apply after version resolution. +4. `SMART_ROUTING_ENV_KEYS` controls environment snapshots, restoration, and launch/session + off overrides. Keep routing activation limited to the V2 and subagent-only flags: + orchestration alone must not enable routing. Add regression tests for the new parameter's + controls. Session overrides must apply after version resolution. `_validate_versions` runs at module import and rejects missing keys, unknown keys, and invalid values. Do not weaken the complete-key check or infer required keys from `_VERSIONS`. diff --git a/src/ucode/constants.py b/src/ucode/constants.py index fbbbdc6a1..46db8fbe2 100644 --- a/src/ucode/constants.py +++ b/src/ucode/constants.py @@ -10,11 +10,9 @@ SMART_ROUTING_ENV_KEYS = ( ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, -) -SMART_ROUTING_CONFIG_ENV_KEYS = ( - *SMART_ROUTING_ENV_KEYS, ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, ) +SMART_ROUTING_CONFIG_ENV_KEYS = SMART_ROUTING_ENV_KEYS MODEL_PROVIDER_SERVICE_HEADER = "Databricks-Model-Provider-Service" MODEL_SERVICE_PARENT_SCHEMA_HEADER = "Databricks-Model-Service-Parent-Schema" diff --git a/src/ucode/smart_routing/v2.py b/src/ucode/smart_routing/v2.py index 4e334849e..6d959e2c6 100644 --- a/src/ucode/smart_routing/v2.py +++ b/src/ucode/smart_routing/v2.py @@ -155,7 +155,9 @@ def smart_routing_enabled( env: MutableMapping[str, str] | None = None, *, default: bool = False ) -> bool: source = resolve_environment(env) - values = [source.get(var) for var in SMART_ROUTING_ENV_KEYS] + values = [ + source.get(var) for var in (ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR) + ] if "1" in values: return True if "0" in values: diff --git a/tests/README.md b/tests/README.md index 366270ef4..bce324318 100644 --- a/tests/README.md +++ b/tests/README.md @@ -137,7 +137,8 @@ PowerShell execution. `test_smart_routing_config.py` covers both canonical v0 selectors against all eight binary legacy-flag combinations, exact precedence/materialization/restoration and getter results, -CLI launch-context materialization/restoration, fail-fast registry validation, +the shared routing/session key registry including the orchestrator exactly once, orchestrator-only +activation defaults, CLI launch-context materialization/restoration, fail-fast registry validation, native-subcommand suppression, and orchestrator state transitions through real temporary session files. These are component checks; they do not establish live agent, hook, or gateway behavior. diff --git a/tests/integration/README.md b/tests/integration/README.md index 71fa8becd..73b47e8b6 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -305,7 +305,8 @@ and execution of pre-existing hooks are not exercised by this integration suite. The unit/component `../test_smart_routing_config.py` covers both canonical-v0 selectors against all eight binary legacy-flag combinations, exact precedence/materialization/restoration and -getter results, CLI launch-context materialization/restoration, registry validation, +getter results, the shared routing/session key registry including the orchestrator exactly once, +orchestrator-only activation defaults, CLI launch-context materialization/restoration, registry validation, process-local environment restoration, native-subcommand suppression, and orchestrator off-to-on transitions through real temporary session files. It does not claim live agent, hook, or gateway coverage. diff --git a/tests/test_smart_router.py b/tests/test_smart_router.py index 81613c882..504cf7991 100644 --- a/tests/test_smart_router.py +++ b/tests/test_smart_router.py @@ -70,7 +70,15 @@ def test_skill_toggles_with_launch_installation_despite_shadowed_path(tmp_path, ) assert result.returncode == 0, result.stdout + result.stderr assert f"Smart Router is {'off' if action == 'disable' else 'on'}" in result.stdout - expected = dict.fromkeys(v2.SMART_ROUTING_ENV_KEYS, "0") if action == "disable" else {} + expected = ( + { + v2.ENABLE_SMART_ROUTING_ENV_VAR: "0", + v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + v2.ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + } + if action == "disable" + else {} + ) assert json.loads(session_path.read_text()) == expected diff --git a/tests/test_smart_routing_config.py b/tests/test_smart_routing_config.py index 96a874bf3..16fed1844 100644 --- a/tests/test_smart_routing_config.py +++ b/tests/test_smart_routing_config.py @@ -19,6 +19,7 @@ ENABLE_SUBAGENT_ROUTING_ENV_VAR, SMART_ROUTING_CONFIG_ENV_KEYS, SMART_ROUTING_CONFIG_VERSION_ENV_VAR, + SMART_ROUTING_ENV_KEYS, ) from ucode.smart_routing import config, orchestrator, session_env, v2 @@ -46,6 +47,16 @@ def _assert_routing_getters(environment: dict[str, str], expected_flags: dict[st ) +def test_smart_routing_key_registry_includes_orchestrator_once(): + assert SMART_ROUTING_ENV_KEYS == ( + ENABLE_SMART_ROUTING_ENV_VAR, + ENABLE_SUBAGENT_ROUTING_ENV_VAR, + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, + ) + assert SMART_ROUTING_ENV_KEYS.count(ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR) == 1 + assert SMART_ROUTING_CONFIG_ENV_KEYS is SMART_ROUTING_ENV_KEYS + + @pytest.mark.parametrize( ("selector", "expected"), [ @@ -194,6 +205,15 @@ def test_legacy_flags_still_drive_routing_queries_without_selector(): assert orchestrator.feature_enabled(environment) is True +@pytest.mark.parametrize("default", [False, True]) +def test_orchestrator_alone_does_not_change_routing_activation_default(default): + environment = {ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1"} + + assert v2.smart_routing_enabled(environment, default=default) is default + assert v2.first_prompt_routing_enabled(environment) is False + assert orchestrator.feature_enabled(environment) is True + + @pytest.mark.parametrize("selector", ["subagent_only_v0", "subagent_orch_v0"]) def test_selector_wins_legacy_conflicts_for_routing_queries(selector): environment = { @@ -310,6 +330,9 @@ def test_config_import_rejects_new_registry_flag_before_runtime_use(monkeypatch) def test_v2_routing_toggles_restore_selector_and_legacy_environment(operation): original = { SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch_v0", + ENABLE_SMART_ROUTING_ENV_VAR: "old-v2", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "old-subagent", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "old-orchestrator", "UNRELATED_SETTING": "preserved", } environment = original.copy() @@ -336,6 +359,7 @@ def test_v2_routing_toggles_restore_selector_and_legacy_environment(operation): assert environment == expected assert SMART_ROUTING_CONFIG_VERSION_ENV_VAR in previous + assert previous[ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR] == "old-orchestrator" v2.restore_smart_routing_env(previous, environment) @@ -351,6 +375,7 @@ def test_explicit_disable_wins_over_selector_until_restored(): assert SMART_ROUTING_CONFIG_VERSION_ENV_VAR not in environment assert environment[ENABLE_SMART_ROUTING_ENV_VAR] == "0" assert environment[ENABLE_SUBAGENT_ROUTING_ENV_VAR] == "0" + assert environment[ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR] == "0" assert v2.smart_routing_enabled(environment) is False assert v2.first_prompt_routing_enabled(environment) is False From 5e0785d630cc4470c560e87f79b03024b6f9b0c2 Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Thu, 8 Oct 2026 21:02:54 +0000 Subject: [PATCH 07/30] Resolve smart routing versions before CLI command processing --- AGENTS.md | 10 +- README.md | 14 +- src/ucode/cli.py | 25 +- src/ucode/constants.py | 1 - src/ucode/smart_routing/config.py | 13 +- tests/README.md | 14 +- tests/integration/README.md | 12 +- tests/test_smart_routing_config.py | 397 ++++++++++++++++++----------- 8 files changed, 318 insertions(+), 168 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index c2c558fd0..205c55a47 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -33,16 +33,20 @@ Tests live in `tests/`. `SMART_ROUTING_CONFIG_VERSION` is the external selector. Version definitions live in `src/ucode/smart_routing/config.py` under `_VERSIONS`; their managed environment keys -are registered in `SMART_ROUTING_CONFIG_ENV_KEYS` in `src/ucode/constants.py`. +are registered in `SMART_ROUTING_ENV_KEYS` in `src/ucode/constants.py`. An unset or empty selector preserves legacy environment-flag behavior. A valid nonempty selector overrides every conflicting legacy value in that registry. Do not use `setdefault` or preserve inherited values for version-owned parameters. +Resolve and materialize the selector at the CLI boundary before argument parsing or callbacks, +including setup, session controls, authentication, and managed-config discovery. Invalid +selectors must fail before those operations. Restore the inherited environment on every exit. Explicit launch/session on/off controls still apply after version expansion. +Managed routing defaults must not rewrite already-resolved version flags. ### Adding a parameter 1. Define its environment-variable constant in `src/ucode/constants.py` and add it to - `SMART_ROUTING_ENV_KEYS`, which also defines `SMART_ROUTING_CONFIG_ENV_KEYS`. + `SMART_ROUTING_ENV_KEYS`. 2. Set an explicit value for it in **every** `_VERSIONS` entry, including existing versions. Choose values that preserve existing versions' behavior. Current parameters accept only the strings `"0"` and `"1"`; do not use booleans, empty strings, or omitted keys. @@ -63,7 +67,7 @@ If a new parameter needs nonbinary values, add parameter-specific validation and 1. Add a complete mapping to `_VERSIONS` with an explicit suffix, such as `new_mode_v0`. For a changed existing mode, add `existing_mode_v1` rather than changing its `_v0` behavior. Do not add unsuffixed names or version aliases. -2. Define every key in `SMART_ROUTING_CONFIG_ENV_KEYS`; never rely on the caller's +2. Define every key in `SMART_ROUTING_ENV_KEYS`; never rely on the caller's inherited environment to fill missing values. 3. Update the version table and examples in `README.md` and any affected bundled-skill docs. 4. Extend `tests/test_smart_routing_config.py` for the new mode, precedence over legacy flags, diff --git a/README.md b/README.md index 12230e4c2..d26c45ad9 100644 --- a/README.md +++ b/README.md @@ -253,8 +253,13 @@ Use `SMART_ROUTING_CONFIG_VERSION` at launch to select a smart-routing configura | Version | Subagent routing | First-prompt routing | Orchestrator | | --- | --- | --- | --- | | `subagent_only_v0` | On | Off | Off | +| `subagent_only_v1` | On | Off | Off | | `subagent_orch_v0` | On | Off | On | +`subagent_only_v1` sets both `ENABLE_SMART_ROUTING_V2` and +`ENABLE_SMART_ROUTING_SUBAGENT_ONLY` to `"1"`. Subagent-only takes precedence, +so first-prompt routing remains off; orchestration is also off. + Smart-routed Claude and Codex sessions install `smart-router`. The `subagent_orch_v0` version also installs and activates the bundled `smart-router-orchestrator` skill. For example: @@ -263,18 +268,21 @@ For example: SMART_ROUTING_CONFIG_VERSION=subagent_orch_v0 ug claude ``` -The version takes precedence over conflicting legacy flags. UG expands it into +The version takes precedence over conflicting legacy flags. Before parsing command options +or running any command callbacks, UG expands it into `ENABLE_SMART_ROUTING_V2`, `ENABLE_SMART_ROUTING_SUBAGENT_ONLY`, and `ENABLE_SMART_ROUTER_ORCHESTRATOR` for the launched session. When the version is unset or empty, these legacy flags retain their existing behavior, including first-prompt routing through `ENABLE_SMART_ROUTING_V2=1`. Unknown versions produce -an error listing the supported values. Orchestration remains off by default. +an error listing the supported values before setup, authentication, or session changes. +Explicit launch/session on/off controls apply after expansion. Orchestration remains off by default. +Workspace smart-routing defaults do not rewrite the selected version's flags. Version names require an explicit suffix. Future revisions use new `_v1`, `_v2`, etc. names without changing existing versions. Version definitions fail validation at module import if any flag in -`SMART_ROUTING_CONFIG_ENV_KEYS` is missing, has a value other than `"0"` or `"1"`, +`SMART_ROUTING_ENV_KEYS` is missing, has a value other than `"0"` or `"1"`, or an unknown flag is present. Register new managed flags in that tuple and explicitly set them in every version. diff --git a/src/ucode/cli.py b/src/ucode/cli.py index ed49c1c13..21fe69b66 100644 --- a/src/ucode/cli.py +++ b/src/ucode/cli.py @@ -1326,6 +1326,25 @@ def revert() -> int: class _HelpOrderedGroup(TyperGroup): """Keep top-level help organized across commands and nested Typer apps.""" + def make_context( + self, + info_name: str | None, + args: list[str], + parent: _click.Context | None = None, + **extra: Any, + ) -> _click.Context: + try: + previous = smart_routing_v2.apply_config() + except RuntimeError as exc: + raise _click.ClickException(str(exc)) from None + try: + ctx = super().make_context(info_name, args, parent, **extra) + except BaseException: + smart_routing_v2.restore_smart_routing_env(previous) + raise + ctx.call_on_close(lambda: smart_routing_v2.restore_smart_routing_env(previous)) + return ctx + def list_commands(self, ctx: _click.Context) -> list[str]: commands = super().list_commands(ctx) order = {name: index for index, name in enumerate(_HELP_COMMAND_ORDER)} @@ -2940,7 +2959,11 @@ def _launch_tool( ) print_success(f"Starting {TOOL_SPECS[tool]['display']}") with _smart_routing_v2_flag( - True if managed_smart_routing_enabled and smart_routing_enabled else None + True + if managed_smart_routing_enabled + and smart_routing_enabled + and not smart_routing_v2.smart_routing_enabled() + else None ): launch_agent(tool, state, ctx.args, options=launch_options) except RuntimeError as exc: diff --git a/src/ucode/constants.py b/src/ucode/constants.py index 46db8fbe2..81fd53f82 100644 --- a/src/ucode/constants.py +++ b/src/ucode/constants.py @@ -12,7 +12,6 @@ ENABLE_SUBAGENT_ROUTING_ENV_VAR, ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, ) -SMART_ROUTING_CONFIG_ENV_KEYS = SMART_ROUTING_ENV_KEYS MODEL_PROVIDER_SERVICE_HEADER = "Databricks-Model-Provider-Service" MODEL_SERVICE_PARENT_SCHEMA_HEADER = "Databricks-Model-Service-Parent-Schema" diff --git a/src/ucode/smart_routing/config.py b/src/ucode/smart_routing/config.py index b8cb26bb8..a7631ad57 100644 --- a/src/ucode/smart_routing/config.py +++ b/src/ucode/smart_routing/config.py @@ -9,8 +9,8 @@ ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, - SMART_ROUTING_CONFIG_ENV_KEYS, SMART_ROUTING_CONFIG_VERSION_ENV_VAR, + SMART_ROUTING_ENV_KEYS, ) _VERSIONS = { @@ -19,6 +19,11 @@ ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", }, + "subagent_only_v1": { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + }, "subagent_orch_v0": { ENABLE_SMART_ROUTING_ENV_VAR: "0", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", @@ -29,7 +34,7 @@ def _validate_versions(versions: Mapping[str, Mapping[str, str]]) -> None: """Require every version to explicitly configure the complete managed flag set.""" - expected = set(SMART_ROUTING_CONFIG_ENV_KEYS) + expected = set(SMART_ROUTING_ENV_KEYS) for version, values in versions.items(): missing = expected - values.keys() unexpected = values.keys() - expected @@ -71,8 +76,8 @@ def apply_config(env: MutableMapping[str, str] | None = None) -> dict[str, str | if not version: return {} resolved = resolve_environment(target) - keys = (*SMART_ROUTING_CONFIG_ENV_KEYS, SMART_ROUTING_CONFIG_VERSION_ENV_VAR) + keys = (*SMART_ROUTING_ENV_KEYS, SMART_ROUTING_CONFIG_VERSION_ENV_VAR) previous = {key: target.get(key) for key in keys} - target.update({key: resolved[key] for key in SMART_ROUTING_CONFIG_ENV_KEYS}) + target.update({key: resolved[key] for key in SMART_ROUTING_ENV_KEYS}) target.pop(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, None) return previous diff --git a/tests/README.md b/tests/README.md index bce324318..766f0e062 100644 --- a/tests/README.md +++ b/tests/README.md @@ -135,13 +135,15 @@ that Claude settings and Codex's shell policy carry the interpreter and session These are component checks; they do not establish native skill permission matching or PowerShell execution. -`test_smart_routing_config.py` covers both canonical v0 selectors against all eight binary -legacy-flag combinations, exact precedence/materialization/restoration and getter results, +`test_smart_routing_config.py` covers all three canonical v0/v1 selectors against every binary +legacy-flag combination, including `subagent_only_v1`'s V2/subagent-on and first-prompt-off +behavior, exact preset precedence/materialization/restoration and getter results, the shared routing/session key registry including the orchestrator exactly once, orchestrator-only -activation defaults, CLI launch-context materialization/restoration, fail-fast registry validation, -native-subcommand suppression, and orchestrator state transitions through real temporary -session files. These are component checks; they do not establish live agent, hook, or gateway -behavior. +activation defaults, CLI startup ordering before callbacks, early invalid-selector exits, +explicit routing controls, fail-fast registry validation, native-subcommand suppression, and +orchestrator state transitions through real temporary session files. Full CLI launch coverage +also verifies that a managed routing opt-in preserves the exact `subagent_only_v0` flags. +These are component checks; they do not establish live agent, hook, or gateway behavior. The toggle integration journeys run with `ENABLE_SMART_ROUTER_ORCHESTRATOR` unset and with `ENABLE_SMART_ROUTER_ORCHESTRATOR=1`. They require only `smart-router` by default and both diff --git a/tests/integration/README.md b/tests/integration/README.md index 73b47e8b6..dd6a203c8 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -303,13 +303,15 @@ role-contract preservation, and isolation from legacy preference files lack dedicated regression coverage. Codex's native hook merging, project trust, and execution of pre-existing hooks are not exercised by this integration suite. -The unit/component `../test_smart_routing_config.py` covers both canonical-v0 selectors against -all eight binary legacy-flag combinations, exact precedence/materialization/restoration and +The unit/component `../test_smart_routing_config.py` covers all three canonical v0/v1 selectors +against every binary legacy-flag combination, including `subagent_only_v1`'s V2/subagent-on and +first-prompt-off behavior, exact preset precedence/materialization/restoration and getter results, the shared routing/session key registry including the orchestrator exactly once, -orchestrator-only activation defaults, CLI launch-context materialization/restoration, registry validation, -process-local environment restoration, +orchestrator-only activation defaults, CLI startup ordering before callbacks, early invalid-selector +exits, explicit routing controls, registry validation, process-local environment restoration, native-subcommand suppression, and orchestrator off-to-on transitions through real temporary -session files. It does not claim live agent, hook, or gateway coverage. +session files. Full CLI launch coverage also verifies that a managed routing opt-in preserves the +exact `subagent_only_v0` flags. It does not claim live agent, hook, or gateway coverage. The portable `../test_claude_windows_smart_routing.py` checks the Windows subagent-only fallback without Unix imports. Native Windows TUI and hook execution diff --git a/tests/test_smart_routing_config.py b/tests/test_smart_routing_config.py index 16fed1844..e1e19d875 100644 --- a/tests/test_smart_routing_config.py +++ b/tests/test_smart_routing_config.py @@ -17,7 +17,6 @@ ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, - SMART_ROUTING_CONFIG_ENV_KEYS, SMART_ROUTING_CONFIG_VERSION_ENV_VAR, SMART_ROUTING_ENV_KEYS, ) @@ -31,6 +30,11 @@ ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", }, + "subagent_only_v1": { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + }, "subagent_orch_v0": { ENABLE_SMART_ROUTING_ENV_VAR: "0", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", @@ -48,55 +52,14 @@ def _assert_routing_getters(environment: dict[str, str], expected_flags: dict[st def test_smart_routing_key_registry_includes_orchestrator_once(): - assert SMART_ROUTING_ENV_KEYS == ( + assert set(SMART_ROUTING_ENV_KEYS) == { ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, - ) - assert SMART_ROUTING_ENV_KEYS.count(ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR) == 1 - assert SMART_ROUTING_CONFIG_ENV_KEYS is SMART_ROUTING_ENV_KEYS - - -@pytest.mark.parametrize( - ("selector", "expected"), - [ - ( - "subagent_only_v0", - { - ENABLE_SMART_ROUTING_ENV_VAR: "0", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", - }, - ), - ( - "subagent_orch_v0", - { - ENABLE_SMART_ROUTING_ENV_VAR: "0", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", - }, - ), - ], -) -def test_resolve_environment_materializes_selector_without_mutating_input(selector, expected): - source = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", - "UNRELATED_SETTING": "preserved", - } - - resolved = config.resolve_environment(source) - - assert resolved == {**expected, "UNRELATED_SETTING": "preserved"} - assert source == { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", - "UNRELATED_SETTING": "preserved", } + assert len(SMART_ROUTING_ENV_KEYS) == 3 + assert len(set(SMART_ROUTING_ENV_KEYS)) == len(SMART_ROUTING_ENV_KEYS) + assert SMART_ROUTING_ENV_KEYS.count(ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR) == 1 @pytest.mark.parametrize( @@ -104,33 +67,36 @@ def test_resolve_environment_materializes_selector_without_mutating_input(select [ (selector, legacy_values, expected_flags) for selector, expected_flags in _SELECTOR_EXPECTED_FLAGS.items() - for legacy_values in product(("0", "1"), repeat=len(SMART_ROUTING_CONFIG_ENV_KEYS)) + for legacy_values in product(("0", "1"), repeat=len(SMART_ROUTING_ENV_KEYS)) ], ) def test_selector_precedence_covers_every_legacy_flag_combination( selector, legacy_values, expected_flags ): - legacy_environment = dict(zip(SMART_ROUTING_CONFIG_ENV_KEYS, legacy_values, strict=True)) + legacy_environment = dict(zip(SMART_ROUTING_ENV_KEYS, legacy_values, strict=True)) original = { SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, **legacy_environment, + "UNRELATED_SETTING": "preserved", } + expected_environment = {**expected_flags, "UNRELATED_SETTING": "preserved"} _assert_routing_getters(original, expected_flags) resolved = config.resolve_environment(original) - assert resolved == expected_flags + assert resolved == expected_environment _assert_routing_getters(resolved, expected_flags) assert original == { SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, **legacy_environment, + "UNRELATED_SETTING": "preserved", } applied = original.copy() previous = config.apply_config(applied) - assert applied == expected_flags + assert applied == expected_environment assert previous == { **legacy_environment, SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, @@ -142,20 +108,6 @@ def test_selector_precedence_covers_every_legacy_flag_combination( assert applied == original -@pytest.mark.parametrize( - ("selector", "expected_orchestrator"), - [("subagent_only_v0", "0"), ("subagent_orch_v0", "1")], -) -def test_resolve_environment_supports_canonical_v0_selectors(selector, expected_orchestrator): - resolved = config.resolve_environment({SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector}) - - assert resolved == { - ENABLE_SMART_ROUTING_ENV_VAR: "0", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: expected_orchestrator, - } - - @pytest.mark.parametrize("selector", ["subagent_only", "subagent_orch"]) @pytest.mark.parametrize("resolver", [config.resolve_environment, config.apply_config]) def test_unsuffixed_selectors_are_rejected_without_mutating_input(selector, resolver): @@ -214,54 +166,16 @@ def test_orchestrator_alone_does_not_change_routing_activation_default(default): assert orchestrator.feature_enabled(environment) is True -@pytest.mark.parametrize("selector", ["subagent_only_v0", "subagent_orch_v0"]) -def test_selector_wins_legacy_conflicts_for_routing_queries(selector): - environment = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", - } - - assert v2.smart_routing_enabled(environment) is True - assert v2.first_prompt_routing_enabled(environment) is False - assert orchestrator.feature_enabled(environment) is (selector == "subagent_orch_v0") - - -def test_apply_config_materializes_selector_and_returns_previous_owned_values(): - environment = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_only_v0", - ENABLE_SMART_ROUTING_ENV_VAR: "old-v2", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "old-subagent", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "old-orchestrator", - "UNRELATED_SETTING": "preserved", - } - - previous = config.apply_config(environment) - - assert set(previous) == { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR, - *SMART_ROUTING_CONFIG_ENV_KEYS, - } - assert previous == { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_only_v0", - ENABLE_SMART_ROUTING_ENV_VAR: "old-v2", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "old-subagent", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "old-orchestrator", - } - assert environment == { - ENABLE_SMART_ROUTING_ENV_VAR: "0", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", - "UNRELATED_SETTING": "preserved", - } +@pytest.mark.parametrize("selector", ["future_mode", "subagent_orch_v1"]) +@pytest.mark.parametrize("resolver", [config.resolve_environment, config.apply_config]) +def test_unknown_selectors_remain_unknown(selector, resolver): + environment = {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector} + with pytest.raises(RuntimeError) as caught: + resolver(environment) -@pytest.mark.parametrize("selector", ["subagent_only_v1", "subagent_orch_v1"]) -@pytest.mark.parametrize("resolver", [config.resolve_environment, config.apply_config]) -def test_future_v1_selectors_remain_unknown(selector, resolver): - with pytest.raises(RuntimeError): - resolver({SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector}) + assert selector in str(caught.value) + assert environment == {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector} @pytest.mark.parametrize("selector", [None, "", " "]) @@ -275,7 +189,7 @@ def test_apply_config_is_a_noop_for_unset_or_blank_selector(selector): assert environment == original -def test_unknown_selector_mentions_supported_names_and_does_not_mutate_input(): +def test_unknown_selector_mentions_supported_names(): for resolver in (config.resolve_environment, config.apply_config): environment = {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "future_mode"} @@ -285,13 +199,14 @@ def test_unknown_selector_mentions_supported_names_and_does_not_mutate_input(): message = str(caught.value) assert "future_mode" in message assert "subagent_only_v0" in message + assert "subagent_only_v1" in message assert "subagent_orch_v0" in message assert environment == {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "future_mode"} -@pytest.mark.parametrize("missing_key", SMART_ROUTING_CONFIG_ENV_KEYS) +@pytest.mark.parametrize("missing_key", SMART_ROUTING_ENV_KEYS) def test_validate_versions_rejects_each_missing_managed_flag(missing_key): - values = {key: "0" for key in SMART_ROUTING_CONFIG_ENV_KEYS if key != missing_key} + values = {key: "0" for key in SMART_ROUTING_ENV_KEYS if key != missing_key} with pytest.raises(ValueError): config._validate_versions({"test_version": values}) @@ -299,7 +214,7 @@ def test_validate_versions_rejects_each_missing_managed_flag(missing_key): @pytest.mark.parametrize("invalid_value", [None, "", "2", "true", 0, False]) def test_validate_versions_rejects_non_binary_managed_flag_values(invalid_value): - values = dict.fromkeys(SMART_ROUTING_CONFIG_ENV_KEYS, "0") + values = dict.fromkeys(SMART_ROUTING_ENV_KEYS, "0") values[ENABLE_SMART_ROUTING_ENV_VAR] = invalid_value with pytest.raises(ValueError): @@ -307,7 +222,7 @@ def test_validate_versions_rejects_non_binary_managed_flag_values(invalid_value) def test_validate_versions_rejects_unexpected_managed_flag(): - values = dict.fromkeys(SMART_ROUTING_CONFIG_ENV_KEYS, "0") + values = dict.fromkeys(SMART_ROUTING_ENV_KEYS, "0") values["UNEXPECTED_SMART_ROUTING_FLAG"] = "0" with pytest.raises(ValueError): @@ -318,8 +233,8 @@ def test_config_import_rejects_new_registry_flag_before_runtime_use(monkeypatch) new_key = "ENABLE_SMART_ROUTING_TEST_ONLY" monkeypatch.setattr( constants, - "SMART_ROUTING_CONFIG_ENV_KEYS", - (*constants.SMART_ROUTING_CONFIG_ENV_KEYS, new_key), + "SMART_ROUTING_ENV_KEYS", + (*constants.SMART_ROUTING_ENV_KEYS, new_key), ) with pytest.raises(ValueError): @@ -406,40 +321,191 @@ def test_cli_context_materializes_inherited_selector_and_restores_after_failure( assert {key: os.environ.get(key) for key in original} == original +def test_cli_context_turns_invalid_selector_into_actionable_exit(monkeypatch): + monkeypatch.setenv(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "future_mode") + + with pytest.raises(typer.Exit) as caught: + with cli._smart_routing_v2_flag(None): + pytest.fail("invalid selector should prevent entering the context") + + assert caught.value.exit_code == 1 + assert os.environ[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] == "future_mode" + + +def test_cli_startup_resolves_selector_before_command_callbacks(monkeypatch): + selector = "subagent_only_v1" + expected_flags = _SELECTOR_EXPECTED_FLAGS[selector] + original = { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + } + for key, value in original.items(): + monkeypatch.setenv(key, value) + observed = [] + + def record_callback(name): + observed.append( + { + "callback": name, + "selector": os.environ.get(SMART_ROUTING_CONFIG_VERSION_ENV_VAR), + "flags": {key: os.environ.get(key) for key in SMART_ROUTING_ENV_KEYS}, + } + ) + + def record_session_toggle(enabled): + assert enabled is None + record_callback("session_toggle") + return False + + def record_custom_oauth(*_arguments, **_options): + record_callback("custom_oauth") + return None + + def record_launch(*_arguments, **_options): + record_callback("launch") + + with ( + patch.object( + cli, + "_toggle_current_smart_routing_session", + side_effect=record_session_toggle, + ), + patch.object(cli, "_custom_oauth_config", side_effect=record_custom_oauth), + patch.object(cli, "_launch_tool", side_effect=record_launch), + ): + result = runner.invoke(cli.app, ["claude", "--client-id", "client"]) + + assert result.exit_code == 0, result.output + assert observed == [ + {"callback": "session_toggle", "selector": None, "flags": expected_flags}, + {"callback": "custom_oauth", "selector": None, "flags": expected_flags}, + {"callback": "launch", "selector": None, "flags": expected_flags}, + ] + assert {key: os.environ.get(key) for key in original} == original + + @pytest.mark.parametrize( - ("selector", "expected_flags"), - list(_SELECTOR_EXPECTED_FLAGS.items()), + "arguments", + [[], ["--version"], ["configure"], ["claude", "--enable-smart-routing"]], + ids=["bare-launch", "version", "configure", "session-toggle"], ) -def test_cli_context_materializes_each_selector_over_opposite_legacy_flags( - monkeypatch, selector, expected_flags -): - legacy_environment = { - key: "1" if value == "0" else "0" for key, value in expected_flags.items() +def test_cli_startup_rejects_invalid_selector_before_callbacks(monkeypatch, arguments): + original = { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "future_mode", + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", } + for key, value in original.items(): + monkeypatch.setenv(key, value) + + with ( + patch("ucode.telemetry.ug_version") as version, + patch.object(cli, "_custom_oauth_config") as custom_oauth, + patch.object(cli, "_toggle_current_smart_routing_session") as session_toggle, + patch.object(cli, "configure_workspace_command") as configure_workspace, + patch.object(cli, "install_databricks_cli") as install_cli, + patch.object(cli, "_launch_managed_default") as launch_managed_default, + ): + result = runner.invoke(cli.app, arguments) + + assert result.exit_code == 1, result.output + assert "Unknown SMART_ROUTING_CONFIG_VERSION" in result.output + for supported_selector in _SELECTOR_EXPECTED_FLAGS: + assert supported_selector in result.output + version.assert_not_called() + custom_oauth.assert_not_called() + session_toggle.assert_not_called() + configure_workspace.assert_not_called() + install_cli.assert_not_called() + launch_managed_default.assert_not_called() + assert {key: os.environ.get(key) for key in original} == original + + +@pytest.mark.parametrize( + ("exit_case", "arguments", "expected_exit_code"), + [ + ("version", ["--version"], 0), + ("command-error", ["claude"], 1), + ("parse-error", ["--workspace"], 2), + ], +) +def test_cli_startup_restores_parent_environment_after_exit( + monkeypatch, exit_case, arguments, expected_exit_code +): original = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, - **legacy_environment, + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_only_v1", + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", } for key, value in original.items(): monkeypatch.setenv(key, value) + launch_error = RuntimeError("launch failed") if exit_case == "command-error" else None - with cli._smart_routing_v2_flag(None): - assert SMART_ROUTING_CONFIG_VERSION_ENV_VAR not in os.environ - assert {key: os.environ.get(key) for key in SMART_ROUTING_CONFIG_ENV_KEYS} == expected_flags - _assert_routing_getters(dict(os.environ), expected_flags) + with patch.object(cli, "_launch_tool", side_effect=launch_error) as launch: + result = runner.invoke(cli.app, arguments) + assert result.exit_code == expected_exit_code, result.output assert {key: os.environ.get(key) for key in original} == original + if exit_case in {"version", "parse-error"}: + launch.assert_not_called() + else: + launch.assert_called_once() -def test_cli_context_turns_invalid_selector_into_actionable_exit(monkeypatch): - monkeypatch.setenv(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "future_mode") +@pytest.mark.parametrize( + ("flag", "expected_flags"), + [ + ( + "--enable-smart-routing", + { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + }, + ), + ( + "--disable-smart-routing", + { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + }, + ), + ], +) +def test_cli_explicit_routing_control_overrides_resolved_selector( + monkeypatch, flag, expected_flags +): + original = { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_only_v0", + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + } + for key, value in original.items(): + monkeypatch.setenv(key, value) + observed = [] - with pytest.raises(typer.Exit) as caught: - with cli._smart_routing_v2_flag(None): - pytest.fail("invalid selector should prevent entering the context") + def record_launch(*_arguments, **_options): + observed.append( + { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: os.environ.get( + SMART_ROUTING_CONFIG_VERSION_ENV_VAR + ), + **{key: os.environ.get(key) for key in SMART_ROUTING_ENV_KEYS}, + } + ) - assert caught.value.exit_code == 1 - assert os.environ[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] == "future_mode" + with patch.object(cli, "_launch_tool", side_effect=record_launch): + result = runner.invoke(cli.app, ["codex", flag]) + + assert result.exit_code == 0, result.output + assert observed == [{SMART_ROUTING_CONFIG_VERSION_ENV_VAR: None, **expected_flags}] + assert {key: os.environ.get(key) for key in original} == original def test_bare_launch_materializes_selector_for_managed_default_and_restores_environment( @@ -483,16 +549,57 @@ def inspect_managed_default(*_args, **_kwargs): assert {key: os.environ.get(key) for key in original} == original -def test_bare_launch_rejects_invalid_selector_before_managed_default(monkeypatch): - monkeypatch.setenv(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "future_mode") +def test_managed_routing_preserves_subagent_only_v0_flags_during_full_launch(monkeypatch): + original = { + SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_only_v0", + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + } + for key, value in original.items(): + monkeypatch.setenv(key, value) + launch_state = { + "workspace": "https://example.databricks.com", + "available_tools": ["claude"], + "base_urls": {"claude": "https://example.databricks.com/ai-gateway/anthropic"}, + "claude_models": {"sonnet": "databricks-claude-sonnet-4"}, + "managed_configs": {}, + } + managed = {"enabled_agents": {"claude": {"smart_routing_enabled": True}}} + observed = [] - with patch.object(cli, "_launch_managed_default") as launch: - result = runner.invoke(cli.app, []) + def record_launch(*_arguments, **_options): + observed.append({key: os.environ.get(key) for key in SMART_ROUTING_ENV_KEYS}) + + with ( + patch.object(cli, "ensure_bootstrap_dependencies"), + patch.object(cli, "load_state", return_value=launch_state), + patch.object(cli, "ensure_provider_state", return_value=launch_state), + patch.object(cli, "_fetch_managed_config", return_value=(managed, False)), + patch.object(cli, "_fetch_budget_recommendation", return_value=None), + patch.object(cli, "get_provider_service", return_value=None), + patch.object(cli, "configure_shared_state", return_value=launch_state), + patch.object( + cli, + "resolve_launch_model", + return_value=(launch_state, "databricks-claude-sonnet-4"), + ), + patch.object(cli, "configure_tool", return_value=launch_state), + patch.object(cli, "refresh_downloaded_skills_on_launch"), + patch.object(cli, "launch_agent", side_effect=record_launch) as launch, + ): + result = runner.invoke(cli.app, ["claude"]) - assert result.exit_code == 1 - launch.assert_not_called() - assert "subagent_only_v0" in result.output - assert "subagent_orch_v0" in result.output + assert result.exit_code == 0, result.output + launch.assert_called_once() + assert observed == [ + { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + } + ] + assert {key: os.environ.get(key) for key in original} == original @pytest.mark.parametrize( From 1aa98801f916d66259a0505d4ed5d58d4d873c34 Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Thu, 8 Oct 2026 21:09:20 +0000 Subject: [PATCH 08/30] Replace smart routing config tests with exhaustive permutation grid --- tests/README.md | 19 +- tests/integration/README.md | 18 +- tests/test_smart_routing_config.py | 708 ++++------------------------- 3 files changed, 112 insertions(+), 633 deletions(-) diff --git a/tests/README.md b/tests/README.md index 766f0e062..94a6f5216 100644 --- a/tests/README.md +++ b/tests/README.md @@ -135,15 +135,16 @@ that Claude settings and Codex's shell policy carry the interpreter and session These are component checks; they do not establish native skill permission matching or PowerShell execution. -`test_smart_routing_config.py` covers all three canonical v0/v1 selectors against every binary -legacy-flag combination, including `subagent_only_v1`'s V2/subagent-on and first-prompt-off -behavior, exact preset precedence/materialization/restoration and getter results, -the shared routing/session key registry including the orchestrator exactly once, orchestrator-only -activation defaults, CLI startup ordering before callbacks, early invalid-selector exits, -explicit routing controls, fail-fast registry validation, native-subcommand suppression, and -orchestrator state transitions through real temporary session files. Full CLI launch coverage -also verifies that a managed routing opt-in preserves the exact `subagent_only_v0` flags. -These are component checks; they do not establish live agent, hook, or gateway behavior. +`test_smart_routing_config.py` is a 768-case Cartesian component oracle: all three legacy +routing flags take unset, blank, `0`, and `1`, while the selector takes unset, blank, +whitespace-only, all three canonical presets, whitespace-padded presets, unsuffixed names, +and an unsupported version. It independently hardcodes each preset and asserts exact +`resolve_environment` and `apply_config` settings, unrelated-key preservation, valid-selector +consumption, blank-selector preservation by `apply_config`, legacy-value preservation, and +nonmutation on invalid input. This file intentionally does not cover CLI startup ordering, +import-time schema validation, snapshots/restoration, routing getters, native subcommands, +hooks, session files, or managed launches; these are not established by this grid. These are +component checks; they do not establish live agent, hook, or gateway behavior. The toggle integration journeys run with `ENABLE_SMART_ROUTER_ORCHESTRATOR` unset and with `ENABLE_SMART_ROUTER_ORCHESTRATOR=1`. They require only `smart-router` by default and both diff --git a/tests/integration/README.md b/tests/integration/README.md index dd6a203c8..59c5f8808 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -303,15 +303,15 @@ role-contract preservation, and isolation from legacy preference files lack dedicated regression coverage. Codex's native hook merging, project trust, and execution of pre-existing hooks are not exercised by this integration suite. -The unit/component `../test_smart_routing_config.py` covers all three canonical v0/v1 selectors -against every binary legacy-flag combination, including `subagent_only_v1`'s V2/subagent-on and -first-prompt-off behavior, exact preset precedence/materialization/restoration and -getter results, the shared routing/session key registry including the orchestrator exactly once, -orchestrator-only activation defaults, CLI startup ordering before callbacks, early invalid-selector -exits, explicit routing controls, registry validation, process-local environment restoration, -native-subcommand suppression, and orchestrator off-to-on transitions through real temporary -session files. Full CLI launch coverage also verifies that a managed routing opt-in preserves the -exact `subagent_only_v0` flags. It does not claim live agent, hook, or gateway coverage. +The unit/component `../test_smart_routing_config.py` is a 768-case Cartesian oracle over all +three legacy routing flags and twelve selector forms: unset, blank, whitespace-only, the three +canonical presets, whitespace-padded presets, unsuffixed names, and an unsupported version. It +independently hardcodes preset values and asserts exact `resolve_environment` and `apply_config` +settings, unrelated-key preservation, valid-selector consumption, blank-selector preservation by +`apply_config`, legacy-value preservation, and nonmutation on invalid input. This file no longer +asserts CLI startup ordering, import-time schema validation, snapshots/restoration, routing +getters, native subcommands, hooks, session files, or managed launches; it does not claim live +agent, hook, or gateway coverage. The portable `../test_claude_windows_smart_routing.py` checks the Windows subagent-only fallback without Unix imports. Native Windows TUI and hook execution diff --git a/tests/test_smart_routing_config.py b/tests/test_smart_routing_config.py index e1e19d875..0c145be31 100644 --- a/tests/test_smart_routing_config.py +++ b/tests/test_smart_routing_config.py @@ -1,30 +1,20 @@ -"""Component coverage for versioned smart-routing configuration.""" +"""Exhaustive component coverage for versioned smart-routing configuration.""" from __future__ import annotations -import os -import runpy from itertools import product -from unittest.mock import patch import pytest -import typer -from typer.testing import CliRunner -import ucode.cli as cli -import ucode.constants as constants from ucode.constants import ( ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, SMART_ROUTING_CONFIG_VERSION_ENV_VAR, - SMART_ROUTING_ENV_KEYS, ) -from ucode.smart_routing import config, orchestrator, session_env, v2 +from ucode.smart_routing import config -runner = CliRunner() - -_SELECTOR_EXPECTED_FLAGS = { +_EXPECTED_PRESETS = { "subagent_only_v0": { ENABLE_SMART_ROUTING_ENV_VAR: "0", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", @@ -42,624 +32,112 @@ }, } +_LEGACY_VALUES = (None, "", "0", "1") +_SELECTOR_CASES = ( + ("selector-unset", None, "legacy", None), + ("selector-blank", "", "legacy", None), + ("selector-whitespace", " \t", "legacy", None), + ("subagent-only-v0", "subagent_only_v0", "preset", _EXPECTED_PRESETS["subagent_only_v0"]), + ("subagent-only-v1", "subagent_only_v1", "preset", _EXPECTED_PRESETS["subagent_only_v1"]), + ("subagent-orch-v0", "subagent_orch_v0", "preset", _EXPECTED_PRESETS["subagent_orch_v0"]), + ( + "padded-subagent-only-v0", + " subagent_only_v0 ", + "preset", + _EXPECTED_PRESETS["subagent_only_v0"], + ), + ( + "padded-subagent-only-v1", + " subagent_only_v1 ", + "preset", + _EXPECTED_PRESETS["subagent_only_v1"], + ), + ( + "padded-subagent-orch-v0", + " subagent_orch_v0 ", + "preset", + _EXPECTED_PRESETS["subagent_orch_v0"], + ), + ("unsuffixed-subagent-only", "subagent_only", "invalid", None), + ("unsuffixed-subagent-orch", "subagent_orch", "invalid", None), + ("unsupported-version", "future_mode", "invalid", None), +) -def _assert_routing_getters(environment: dict[str, str], expected_flags: dict[str, str]) -> None: - assert v2.smart_routing_enabled(environment) is True - assert v2.first_prompt_routing_enabled(environment) is False - assert orchestrator.feature_enabled(environment) is ( - expected_flags[ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR] == "1" +_GRID = [ + pytest.param( + v2_value, + subagent_value, + orchestrator_value, + selector, + selector_mode, + expected_flags, + id=( + f"{selector_id}-v2={v2_value!r}-subagent={subagent_value!r}-" + f"orchestrator={orchestrator_value!r}" + ), ) - - -def test_smart_routing_key_registry_includes_orchestrator_once(): - assert set(SMART_ROUTING_ENV_KEYS) == { - ENABLE_SMART_ROUTING_ENV_VAR, - ENABLE_SUBAGENT_ROUTING_ENV_VAR, - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, - } - assert len(SMART_ROUTING_ENV_KEYS) == 3 - assert len(set(SMART_ROUTING_ENV_KEYS)) == len(SMART_ROUTING_ENV_KEYS) - assert SMART_ROUTING_ENV_KEYS.count(ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR) == 1 + for selector_id, selector, selector_mode, expected_flags in _SELECTOR_CASES + for v2_value, subagent_value, orchestrator_value in product(_LEGACY_VALUES, repeat=3) +] @pytest.mark.parametrize( - ("selector", "legacy_values", "expected_flags"), - [ - (selector, legacy_values, expected_flags) - for selector, expected_flags in _SELECTOR_EXPECTED_FLAGS.items() - for legacy_values in product(("0", "1"), repeat=len(SMART_ROUTING_ENV_KEYS)) - ], + ( + "v2_value", + "subagent_value", + "orchestrator_value", + "selector", + "selector_mode", + "expected_flags", + ), + _GRID, ) -def test_selector_precedence_covers_every_legacy_flag_combination( - selector, legacy_values, expected_flags +def test_smart_routing_config_cartesian_grid( + v2_value, + subagent_value, + orchestrator_value, + selector, + selector_mode, + expected_flags, ): - legacy_environment = dict(zip(SMART_ROUTING_ENV_KEYS, legacy_values, strict=True)) - original = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, - **legacy_environment, - "UNRELATED_SETTING": "preserved", - } - expected_environment = {**expected_flags, "UNRELATED_SETTING": "preserved"} - - _assert_routing_getters(original, expected_flags) - - resolved = config.resolve_environment(original) - - assert resolved == expected_environment - _assert_routing_getters(resolved, expected_flags) - assert original == { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, - **legacy_environment, - "UNRELATED_SETTING": "preserved", - } - - applied = original.copy() - previous = config.apply_config(applied) - - assert applied == expected_environment - assert previous == { - **legacy_environment, - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, - } - _assert_routing_getters(applied, expected_flags) - - v2.restore_smart_routing_env(previous, applied) - - assert applied == original - - -@pytest.mark.parametrize("selector", ["subagent_only", "subagent_orch"]) -@pytest.mark.parametrize("resolver", [config.resolve_environment, config.apply_config]) -def test_unsuffixed_selectors_are_rejected_without_mutating_input(selector, resolver): - environment = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, - "UNRELATED_SETTING": "preserved", - } - original = environment.copy() - - with pytest.raises(RuntimeError): - resolver(environment) - - assert environment == original - - -@pytest.mark.parametrize("selector", [None, "", " \t"]) -def test_resolve_environment_uses_legacy_flags_for_blank_or_unset_selector(selector): source = { - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + environment_key: value + for environment_key, value in ( + (ENABLE_SMART_ROUTING_ENV_VAR, v2_value), + (ENABLE_SUBAGENT_ROUTING_ENV_VAR, subagent_value), + (ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, orchestrator_value), + ) + if value is not None } + source["UNRELATED_SETTING"] = "preserved" if selector is not None: source[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] = selector original = source.copy() - resolved = config.resolve_environment(source) - - assert resolved == { - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", - } - assert source == original - - -def test_legacy_flags_still_drive_routing_queries_without_selector(): - environment = { - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", - } - - assert config.resolve_environment(environment) == environment - assert v2.smart_routing_enabled(environment) is True - assert v2.first_prompt_routing_enabled(environment) is True - assert orchestrator.feature_enabled(environment) is True - - -@pytest.mark.parametrize("default", [False, True]) -def test_orchestrator_alone_does_not_change_routing_activation_default(default): - environment = {ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1"} - - assert v2.smart_routing_enabled(environment, default=default) is default - assert v2.first_prompt_routing_enabled(environment) is False - assert orchestrator.feature_enabled(environment) is True - - -@pytest.mark.parametrize("selector", ["future_mode", "subagent_orch_v1"]) -@pytest.mark.parametrize("resolver", [config.resolve_environment, config.apply_config]) -def test_unknown_selectors_remain_unknown(selector, resolver): - environment = {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector} - - with pytest.raises(RuntimeError) as caught: - resolver(environment) - - assert selector in str(caught.value) - assert environment == {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector} - - -@pytest.mark.parametrize("selector", [None, "", " "]) -def test_apply_config_is_a_noop_for_unset_or_blank_selector(selector): - environment = {"UNRELATED_SETTING": "preserved"} - if selector is not None: - environment[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] = selector - original = environment.copy() - - assert config.apply_config(environment) == {} - assert environment == original - - -def test_unknown_selector_mentions_supported_names(): - for resolver in (config.resolve_environment, config.apply_config): - environment = {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "future_mode"} - - with pytest.raises(RuntimeError) as caught: - resolver(environment) - - message = str(caught.value) - assert "future_mode" in message - assert "subagent_only_v0" in message - assert "subagent_only_v1" in message - assert "subagent_orch_v0" in message - assert environment == {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "future_mode"} - - -@pytest.mark.parametrize("missing_key", SMART_ROUTING_ENV_KEYS) -def test_validate_versions_rejects_each_missing_managed_flag(missing_key): - values = {key: "0" for key in SMART_ROUTING_ENV_KEYS if key != missing_key} - - with pytest.raises(ValueError): - config._validate_versions({"test_version": values}) - - -@pytest.mark.parametrize("invalid_value", [None, "", "2", "true", 0, False]) -def test_validate_versions_rejects_non_binary_managed_flag_values(invalid_value): - values = dict.fromkeys(SMART_ROUTING_ENV_KEYS, "0") - values[ENABLE_SMART_ROUTING_ENV_VAR] = invalid_value - - with pytest.raises(ValueError): - config._validate_versions({"test_version": values}) - - -def test_validate_versions_rejects_unexpected_managed_flag(): - values = dict.fromkeys(SMART_ROUTING_ENV_KEYS, "0") - values["UNEXPECTED_SMART_ROUTING_FLAG"] = "0" - - with pytest.raises(ValueError): - config._validate_versions({"test_version": values}) - - -def test_config_import_rejects_new_registry_flag_before_runtime_use(monkeypatch): - new_key = "ENABLE_SMART_ROUTING_TEST_ONLY" - monkeypatch.setattr( - constants, - "SMART_ROUTING_ENV_KEYS", - (*constants.SMART_ROUTING_ENV_KEYS, new_key), - ) - - with pytest.raises(ValueError): - runpy.run_path(config.__file__) - - -@pytest.mark.parametrize("operation", ["enable", "override", "disable"]) -def test_v2_routing_toggles_restore_selector_and_legacy_environment(operation): - original = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch_v0", - ENABLE_SMART_ROUTING_ENV_VAR: "old-v2", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "old-subagent", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "old-orchestrator", - "UNRELATED_SETTING": "preserved", - } - environment = original.copy() - - if operation == "enable": - previous = v2.enable_smart_routing(environment) - expected = { - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", - "UNRELATED_SETTING": "preserved", - } - elif operation == "override": - previous = v2.override_smart_routing(True, environment) - expected = { - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", - "UNRELATED_SETTING": "preserved", - } - else: - previous = v2.disable_smart_routing(environment) - expected = {"UNRELATED_SETTING": "preserved"} - - assert environment == expected - assert SMART_ROUTING_CONFIG_VERSION_ENV_VAR in previous - assert previous[ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR] == "old-orchestrator" - - v2.restore_smart_routing_env(previous, environment) - - assert environment == original - - -def test_explicit_disable_wins_over_selector_until_restored(): - original = {SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch_v0"} - environment = original.copy() - - previous = v2.override_smart_routing(False, environment) - - assert SMART_ROUTING_CONFIG_VERSION_ENV_VAR not in environment - assert environment[ENABLE_SMART_ROUTING_ENV_VAR] == "0" - assert environment[ENABLE_SUBAGENT_ROUTING_ENV_VAR] == "0" - assert environment[ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR] == "0" - assert v2.smart_routing_enabled(environment) is False - assert v2.first_prompt_routing_enabled(environment) is False - - v2.restore_smart_routing_env(previous, environment) - - assert environment == original - - -def test_cli_context_materializes_inherited_selector_and_restores_after_failure(monkeypatch): - original = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch_v0", - ENABLE_SMART_ROUTING_ENV_VAR: "old-v2", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "old-subagent", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "old-orchestrator", - } - for key, value in original.items(): - monkeypatch.setenv(key, value) - - with pytest.raises(ValueError, match="launch failed"): - with cli._smart_routing_v2_flag(None): - assert SMART_ROUTING_CONFIG_VERSION_ENV_VAR not in os.environ - assert os.environ[ENABLE_SMART_ROUTING_ENV_VAR] == "0" - assert os.environ[ENABLE_SUBAGENT_ROUTING_ENV_VAR] == "1" - assert os.environ[ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR] == "1" - assert v2.smart_routing_enabled() is True - raise ValueError("launch failed") + if selector_mode == "invalid": + with pytest.raises(RuntimeError): + config.resolve_environment(source) + assert source == original - assert {key: os.environ.get(key) for key in original} == original + applied = source.copy() + with pytest.raises(RuntimeError): + config.apply_config(applied) + assert applied == original + return + expected_legacy = original.copy() + expected_legacy.pop(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, None) + expected_resolved = expected_legacy.copy() + if selector_mode == "preset": + expected_resolved.update(expected_flags) -def test_cli_context_turns_invalid_selector_into_actionable_exit(monkeypatch): - monkeypatch.setenv(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "future_mode") - - with pytest.raises(typer.Exit) as caught: - with cli._smart_routing_v2_flag(None): - pytest.fail("invalid selector should prevent entering the context") - - assert caught.value.exit_code == 1 - assert os.environ[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] == "future_mode" - - -def test_cli_startup_resolves_selector_before_command_callbacks(monkeypatch): - selector = "subagent_only_v1" - expected_flags = _SELECTOR_EXPECTED_FLAGS[selector] - original = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: selector, - ENABLE_SMART_ROUTING_ENV_VAR: "0", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", - } - for key, value in original.items(): - monkeypatch.setenv(key, value) - observed = [] - - def record_callback(name): - observed.append( - { - "callback": name, - "selector": os.environ.get(SMART_ROUTING_CONFIG_VERSION_ENV_VAR), - "flags": {key: os.environ.get(key) for key in SMART_ROUTING_ENV_KEYS}, - } - ) - - def record_session_toggle(enabled): - assert enabled is None - record_callback("session_toggle") - return False - - def record_custom_oauth(*_arguments, **_options): - record_callback("custom_oauth") - return None - - def record_launch(*_arguments, **_options): - record_callback("launch") - - with ( - patch.object( - cli, - "_toggle_current_smart_routing_session", - side_effect=record_session_toggle, - ), - patch.object(cli, "_custom_oauth_config", side_effect=record_custom_oauth), - patch.object(cli, "_launch_tool", side_effect=record_launch), - ): - result = runner.invoke(cli.app, ["claude", "--client-id", "client"]) - - assert result.exit_code == 0, result.output - assert observed == [ - {"callback": "session_toggle", "selector": None, "flags": expected_flags}, - {"callback": "custom_oauth", "selector": None, "flags": expected_flags}, - {"callback": "launch", "selector": None, "flags": expected_flags}, - ] - assert {key: os.environ.get(key) for key in original} == original - - -@pytest.mark.parametrize( - "arguments", - [[], ["--version"], ["configure"], ["claude", "--enable-smart-routing"]], - ids=["bare-launch", "version", "configure", "session-toggle"], -) -def test_cli_startup_rejects_invalid_selector_before_callbacks(monkeypatch, arguments): - original = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "future_mode", - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", - } - for key, value in original.items(): - monkeypatch.setenv(key, value) - - with ( - patch("ucode.telemetry.ug_version") as version, - patch.object(cli, "_custom_oauth_config") as custom_oauth, - patch.object(cli, "_toggle_current_smart_routing_session") as session_toggle, - patch.object(cli, "configure_workspace_command") as configure_workspace, - patch.object(cli, "install_databricks_cli") as install_cli, - patch.object(cli, "_launch_managed_default") as launch_managed_default, - ): - result = runner.invoke(cli.app, arguments) - - assert result.exit_code == 1, result.output - assert "Unknown SMART_ROUTING_CONFIG_VERSION" in result.output - for supported_selector in _SELECTOR_EXPECTED_FLAGS: - assert supported_selector in result.output - version.assert_not_called() - custom_oauth.assert_not_called() - session_toggle.assert_not_called() - configure_workspace.assert_not_called() - install_cli.assert_not_called() - launch_managed_default.assert_not_called() - assert {key: os.environ.get(key) for key in original} == original - - -@pytest.mark.parametrize( - ("exit_case", "arguments", "expected_exit_code"), - [ - ("version", ["--version"], 0), - ("command-error", ["claude"], 1), - ("parse-error", ["--workspace"], 2), - ], -) -def test_cli_startup_restores_parent_environment_after_exit( - monkeypatch, exit_case, arguments, expected_exit_code -): - original = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_only_v1", - ENABLE_SMART_ROUTING_ENV_VAR: "0", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", - } - for key, value in original.items(): - monkeypatch.setenv(key, value) - launch_error = RuntimeError("launch failed") if exit_case == "command-error" else None - - with patch.object(cli, "_launch_tool", side_effect=launch_error) as launch: - result = runner.invoke(cli.app, arguments) - - assert result.exit_code == expected_exit_code, result.output - assert {key: os.environ.get(key) for key in original} == original - if exit_case in {"version", "parse-error"}: - launch.assert_not_called() - else: - launch.assert_called_once() - - -@pytest.mark.parametrize( - ("flag", "expected_flags"), - [ - ( - "--enable-smart-routing", - { - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", - }, - ), - ( - "--disable-smart-routing", - { - ENABLE_SMART_ROUTING_ENV_VAR: "0", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", - }, - ), - ], -) -def test_cli_explicit_routing_control_overrides_resolved_selector( - monkeypatch, flag, expected_flags -): - original = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_only_v0", - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", - } - for key, value in original.items(): - monkeypatch.setenv(key, value) - observed = [] - - def record_launch(*_arguments, **_options): - observed.append( - { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: os.environ.get( - SMART_ROUTING_CONFIG_VERSION_ENV_VAR - ), - **{key: os.environ.get(key) for key in SMART_ROUTING_ENV_KEYS}, - } - ) - - with patch.object(cli, "_launch_tool", side_effect=record_launch): - result = runner.invoke(cli.app, ["codex", flag]) - - assert result.exit_code == 0, result.output - assert observed == [{SMART_ROUTING_CONFIG_VERSION_ENV_VAR: None, **expected_flags}] - assert {key: os.environ.get(key) for key in original} == original - - -def test_bare_launch_materializes_selector_for_managed_default_and_restores_environment( - monkeypatch, -): - original = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch_v0", - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", - } - for key, value in original.items(): - monkeypatch.setenv(key, value) - observed = [] - - def inspect_managed_default(*_args, **_kwargs): - observed.append( - { - "selector": os.environ.get(SMART_ROUTING_CONFIG_VERSION_ENV_VAR), - "v2": os.environ.get(ENABLE_SMART_ROUTING_ENV_VAR), - "subagent": os.environ.get(ENABLE_SUBAGENT_ROUTING_ENV_VAR), - "orchestrator": os.environ.get(ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR), - } - ) - - with patch.object( - cli, "_launch_managed_default", side_effect=inspect_managed_default - ) as launch: - result = runner.invoke(cli.app, []) - - assert result.exit_code == 0, result.output - launch.assert_called_once() - assert observed == [ - { - "selector": None, - "v2": "0", - "subagent": "1", - "orchestrator": "1", - } - ] - assert {key: os.environ.get(key) for key in original} == original - - -def test_managed_routing_preserves_subagent_only_v0_flags_during_full_launch(monkeypatch): - original = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_only_v0", - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", - } - for key, value in original.items(): - monkeypatch.setenv(key, value) - launch_state = { - "workspace": "https://example.databricks.com", - "available_tools": ["claude"], - "base_urls": {"claude": "https://example.databricks.com/ai-gateway/anthropic"}, - "claude_models": {"sonnet": "databricks-claude-sonnet-4"}, - "managed_configs": {}, - } - managed = {"enabled_agents": {"claude": {"smart_routing_enabled": True}}} - observed = [] - - def record_launch(*_arguments, **_options): - observed.append({key: os.environ.get(key) for key in SMART_ROUTING_ENV_KEYS}) - - with ( - patch.object(cli, "ensure_bootstrap_dependencies"), - patch.object(cli, "load_state", return_value=launch_state), - patch.object(cli, "ensure_provider_state", return_value=launch_state), - patch.object(cli, "_fetch_managed_config", return_value=(managed, False)), - patch.object(cli, "_fetch_budget_recommendation", return_value=None), - patch.object(cli, "get_provider_service", return_value=None), - patch.object(cli, "configure_shared_state", return_value=launch_state), - patch.object( - cli, - "resolve_launch_model", - return_value=(launch_state, "databricks-claude-sonnet-4"), - ), - patch.object(cli, "configure_tool", return_value=launch_state), - patch.object(cli, "refresh_downloaded_skills_on_launch"), - patch.object(cli, "launch_agent", side_effect=record_launch) as launch, - ): - result = runner.invoke(cli.app, ["claude"]) - - assert result.exit_code == 0, result.output - launch.assert_called_once() - assert observed == [ - { - ENABLE_SMART_ROUTING_ENV_VAR: "0", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", - } - ] - assert {key: os.environ.get(key) for key in original} == original - - -@pytest.mark.parametrize( - ("tool", "subcommand"), - [("codex", "app"), ("claude", "update")], -) -def test_native_subcommand_suppresses_inherited_selector_routing(monkeypatch, tool, subcommand): - monkeypatch.setenv(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "subagent_orch_v0") - observed = [] - - with patch( - "ucode.cli._launch_tool", - side_effect=lambda *_args, **_kwargs: observed.append( - { - "selector": os.environ.get(SMART_ROUTING_CONFIG_VERSION_ENV_VAR), - "v2": os.environ.get(ENABLE_SMART_ROUTING_ENV_VAR), - "subagent": os.environ.get(ENABLE_SUBAGENT_ROUTING_ENV_VAR), - "enabled": v2.smart_routing_enabled(), - } - ), - ): - result = runner.invoke(cli.app, [tool, subcommand]) - - assert result.exit_code == 0, result.output - assert observed == [{"selector": None, "v2": None, "subagent": None, "enabled": False}] - assert os.environ[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] == "subagent_orch_v0" - - -def test_session_file_overrides_resolved_selector_for_off_to_on_orchestration( - tmp_path, monkeypatch -): - session_path = tmp_path / "session-env.json" - session_path.write_text("{}", encoding="utf-8") - environment = { - SMART_ROUTING_CONFIG_VERSION_ENV_VAR: "subagent_orch_v0", - session_env.SESSION_ENV_VAR: str(session_path), - } - monkeypatch.setenv(session_env.SESSION_ENV_VAR, str(session_path)) - - session_env.set_session_environment( - { - ENABLE_SMART_ROUTING_ENV_VAR: "0", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - } - ) - off = session_env.effective_environment(environment) + resolved = config.resolve_environment(source) - assert off[ENABLE_SMART_ROUTING_ENV_VAR] == "0" - assert off[ENABLE_SUBAGENT_ROUTING_ENV_VAR] == "0" - assert off[ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR] == "1" - assert v2.smart_routing_enabled(off) is False - assert orchestrator.enabled(environment) is False + assert resolved == expected_resolved + assert source == original - session_env.set_session_environment({}) - on = session_env.effective_environment(environment) + applied = source.copy() + config.apply_config(applied) - assert on[ENABLE_SMART_ROUTING_ENV_VAR] == "0" - assert on[ENABLE_SUBAGENT_ROUTING_ENV_VAR] == "1" - assert on[ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR] == "1" - assert v2.smart_routing_enabled(on) is True - assert v2.first_prompt_routing_enabled(on) is False - assert orchestrator.enabled(environment) is True - assert environment[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] == "subagent_orch_v0" + expected_applied = expected_resolved if selector_mode == "preset" else original + assert applied == expected_applied From 98e41021ed491c259bb24d3c865b1f82257a9ef4 Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Thu, 8 Oct 2026 21:12:46 +0000 Subject: [PATCH 09/30] Simplify smart routing configuration permutation test --- tests/test_smart_routing_config.py | 102 ++++++++--------------------- 1 file changed, 28 insertions(+), 74 deletions(-) diff --git a/tests/test_smart_routing_config.py b/tests/test_smart_routing_config.py index 0c145be31..2a2c4462f 100644 --- a/tests/test_smart_routing_config.py +++ b/tests/test_smart_routing_config.py @@ -2,8 +2,6 @@ from __future__ import annotations -from itertools import product - import pytest from ucode.constants import ( @@ -32,73 +30,29 @@ }, } -_LEGACY_VALUES = (None, "", "0", "1") -_SELECTOR_CASES = ( - ("selector-unset", None, "legacy", None), - ("selector-blank", "", "legacy", None), - ("selector-whitespace", " \t", "legacy", None), - ("subagent-only-v0", "subagent_only_v0", "preset", _EXPECTED_PRESETS["subagent_only_v0"]), - ("subagent-only-v1", "subagent_only_v1", "preset", _EXPECTED_PRESETS["subagent_only_v1"]), - ("subagent-orch-v0", "subagent_orch_v0", "preset", _EXPECTED_PRESETS["subagent_orch_v0"]), - ( - "padded-subagent-only-v0", + +@pytest.mark.parametrize("v2_value", [None, "", "0", "1"]) +@pytest.mark.parametrize("subagent_value", [None, "", "0", "1"]) +@pytest.mark.parametrize("orchestrator_value", [None, "", "0", "1"]) +@pytest.mark.parametrize( + "config_version", + [ + None, + "", + " \t", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", " subagent_only_v0 ", - "preset", - _EXPECTED_PRESETS["subagent_only_v0"], - ), - ( - "padded-subagent-only-v1", " subagent_only_v1 ", - "preset", - _EXPECTED_PRESETS["subagent_only_v1"], - ), - ( - "padded-subagent-orch-v0", " subagent_orch_v0 ", - "preset", - _EXPECTED_PRESETS["subagent_orch_v0"], - ), - ("unsuffixed-subagent-only", "subagent_only", "invalid", None), - ("unsuffixed-subagent-orch", "subagent_orch", "invalid", None), - ("unsupported-version", "future_mode", "invalid", None), -) - -_GRID = [ - pytest.param( - v2_value, - subagent_value, - orchestrator_value, - selector, - selector_mode, - expected_flags, - id=( - f"{selector_id}-v2={v2_value!r}-subagent={subagent_value!r}-" - f"orchestrator={orchestrator_value!r}" - ), - ) - for selector_id, selector, selector_mode, expected_flags in _SELECTOR_CASES - for v2_value, subagent_value, orchestrator_value in product(_LEGACY_VALUES, repeat=3) -] - - -@pytest.mark.parametrize( - ( - "v2_value", - "subagent_value", - "orchestrator_value", - "selector", - "selector_mode", - "expected_flags", - ), - _GRID, + "subagent_only", + "subagent_orch", + "future_mode", + ], ) def test_smart_routing_config_cartesian_grid( - v2_value, - subagent_value, - orchestrator_value, - selector, - selector_mode, - expected_flags, + v2_value, subagent_value, orchestrator_value, config_version ): source = { environment_key: value @@ -110,11 +64,12 @@ def test_smart_routing_config_cartesian_grid( if value is not None } source["UNRELATED_SETTING"] = "preserved" - if selector is not None: - source[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] = selector + if config_version is not None: + source[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] = config_version original = source.copy() + normalized_version = (config_version or "").strip() - if selector_mode == "invalid": + if normalized_version and normalized_version not in _EXPECTED_PRESETS: with pytest.raises(RuntimeError): config.resolve_environment(source) assert source == original @@ -125,19 +80,18 @@ def test_smart_routing_config_cartesian_grid( assert applied == original return - expected_legacy = original.copy() - expected_legacy.pop(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, None) - expected_resolved = expected_legacy.copy() - if selector_mode == "preset": - expected_resolved.update(expected_flags) + expected = original.copy() + expected.pop(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, None) + if normalized_version: + expected.update(_EXPECTED_PRESETS[normalized_version]) resolved = config.resolve_environment(source) - assert resolved == expected_resolved + assert resolved == expected assert source == original applied = source.copy() config.apply_config(applied) - expected_applied = expected_resolved if selector_mode == "preset" else original + expected_applied = expected if normalized_version else original assert applied == expected_applied From c4ffbbba201e60b0d7cd4fbdb90991fedd10e91f Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Thu, 8 Oct 2026 21:30:26 +0000 Subject: [PATCH 10/30] Rename smart router selector and add customer routing preset --- AGENTS.md | 2 +- README.md | 9 +- skills/smart-router-orchestrator/README.md | 4 +- src/ucode/constants.py | 2 +- src/ucode/smart_routing/config.py | 17 ++- src/ucode/smart_routing/v2.py | 8 +- tests/README.md | 56 +++---- tests/conftest.py | 2 +- tests/integration/README.md | 52 +++---- tests/integration/test_ug_claude_commands.py | 50 ++++++- tests/integration/test_ug_claude_headless.py | 23 ++- tests/integration/test_ug_codex_app_server.py | 27 +++- tests/integration/test_ug_codex_commands.py | 127 +++++++++++++--- tests/integration/test_ug_codex_headless.py | 23 ++- .../test_ug_configure_managed_models.py | 94 +++++++++--- .../test_ug_smart_routing_hooks.py | 141 ++++++++++++------ tests/test_smart_routing_config.py | 62 ++++---- 17 files changed, 489 insertions(+), 210 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 205c55a47..66c35a81b 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -31,7 +31,7 @@ Tests live in `tests/`. ## Smart-routing configuration -`SMART_ROUTING_CONFIG_VERSION` is the external selector. Version definitions live in +`SMART_ROUTER_CONFIG_VERSION` is the external selector. Version definitions live in `src/ucode/smart_routing/config.py` under `_VERSIONS`; their managed environment keys are registered in `SMART_ROUTING_ENV_KEYS` in `src/ucode/constants.py`. An unset or empty selector preserves legacy environment-flag behavior. diff --git a/README.md b/README.md index d26c45ad9..1c877db49 100644 --- a/README.md +++ b/README.md @@ -248,14 +248,19 @@ The generated shell hooks expect Git Bash; PowerShell-only setups are not covere ### Smart Router Orchestrator -Use `SMART_ROUTING_CONFIG_VERSION` at launch to select a smart-routing configuration: +Use `SMART_ROUTER_CONFIG_VERSION` at launch to select a smart-routing configuration: | Version | Subagent routing | First-prompt routing | Orchestrator | | --- | --- | --- | --- | +| `first_prompt_and_subagent_no_orch_v0` | On | On | Off | | `subagent_only_v0` | On | Off | Off | | `subagent_only_v1` | On | Off | Off | | `subagent_orch_v0` | On | Off | On | +`first_prompt_and_subagent_no_orch_v0` is the customer configuration for first-prompt +and subagent routing without orchestration: `ENABLE_SMART_ROUTING_V2=1`, +`ENABLE_SMART_ROUTING_SUBAGENT_ONLY=0`, and `ENABLE_SMART_ROUTER_ORCHESTRATOR=0`. + `subagent_only_v1` sets both `ENABLE_SMART_ROUTING_V2` and `ENABLE_SMART_ROUTING_SUBAGENT_ONLY` to `"1"`. Subagent-only takes precedence, so first-prompt routing remains off; orchestration is also off. @@ -265,7 +270,7 @@ version also installs and activates the bundled `smart-router-orchestrator` skil For example: ```bash -SMART_ROUTING_CONFIG_VERSION=subagent_orch_v0 ug claude +SMART_ROUTER_CONFIG_VERSION=subagent_orch_v0 ug claude ``` The version takes precedence over conflicting legacy flags. Before parsing command options diff --git a/skills/smart-router-orchestrator/README.md b/skills/smart-router-orchestrator/README.md index 32d30bdd8..081be4f6e 100644 --- a/skills/smart-router-orchestrator/README.md +++ b/skills/smart-router-orchestrator/README.md @@ -2,10 +2,10 @@ UG bundles the `smart-router-orchestrator` workflow and five Claude role definitions. Smart-routed Claude and Codex launches install and activate this skill alongside -`smart-router` with `SMART_ROUTING_CONFIG_VERSION=subagent_orch_v0`. +`smart-router` with `SMART_ROUTER_CONFIG_VERSION=subagent_orch_v0`. UG expands that version into the session's legacy feature flags. The existing `ENABLE_SMART_ROUTER_ORCHESTRATOR=1` opt-in remains supported when -`SMART_ROUTING_CONFIG_VERSION` is unset; `subagent_only_v0` explicitly leaves orchestration off. +`SMART_ROUTER_CONFIG_VERSION` is unset; `subagent_only_v0` explicitly leaves orchestration off. The feature is off by default; routing alone installs only `smart-router`. The workflow is injected before root prompts and after compaction. The hook checks diff --git a/src/ucode/constants.py b/src/ucode/constants.py index 81fd53f82..030be1aa8 100644 --- a/src/ucode/constants.py +++ b/src/ucode/constants.py @@ -6,7 +6,7 @@ ENABLE_SMART_ROUTING_ENV_VAR = "ENABLE_SMART_ROUTING_V2" ENABLE_SUBAGENT_ROUTING_ENV_VAR = "ENABLE_SMART_ROUTING_SUBAGENT_ONLY" ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR = "ENABLE_SMART_ROUTER_ORCHESTRATOR" -SMART_ROUTING_CONFIG_VERSION_ENV_VAR = "SMART_ROUTING_CONFIG_VERSION" +SMART_ROUTER_CONFIG_VERSION_ENV_VAR = "SMART_ROUTER_CONFIG_VERSION" SMART_ROUTING_ENV_KEYS = ( ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, diff --git a/src/ucode/smart_routing/config.py b/src/ucode/smart_routing/config.py index a7631ad57..d64de93f7 100644 --- a/src/ucode/smart_routing/config.py +++ b/src/ucode/smart_routing/config.py @@ -9,11 +9,16 @@ ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, - SMART_ROUTING_CONFIG_VERSION_ENV_VAR, + SMART_ROUTER_CONFIG_VERSION_ENV_VAR, SMART_ROUTING_ENV_KEYS, ) _VERSIONS = { + "first_prompt_and_subagent_no_orch_v0": { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + }, "subagent_only_v0": { ENABLE_SMART_ROUTING_ENV_VAR: "0", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", @@ -57,12 +62,12 @@ def _validate_versions(versions: Mapping[str, Mapping[str, str]]) -> None: def resolve_environment(env: Mapping[str, str] | None = None) -> dict[str, str]: """Expand a version before applying any launch or session-specific overrides.""" resolved = dict(os.environ if env is None else env) - version = resolved.pop(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "").strip() + version = resolved.pop(SMART_ROUTER_CONFIG_VERSION_ENV_VAR, "").strip() if not version: return resolved if version not in _VERSIONS: raise RuntimeError( - f"Unknown {SMART_ROUTING_CONFIG_VERSION_ENV_VAR} value {version!r}. " + f"Unknown {SMART_ROUTER_CONFIG_VERSION_ENV_VAR} value {version!r}. " f"Use one of: {', '.join(_VERSIONS)}, or unset it to use the legacy flags." ) resolved.update(_VERSIONS[version]) @@ -72,12 +77,12 @@ def resolve_environment(env: Mapping[str, str] | None = None) -> dict[str, str]: def apply_config(env: MutableMapping[str, str] | None = None) -> dict[str, str | None]: """Consume the launch selector, returning the values needed to restore its input.""" target = os.environ if env is None else env - version = target.get(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, "").strip() + version = target.get(SMART_ROUTER_CONFIG_VERSION_ENV_VAR, "").strip() if not version: return {} resolved = resolve_environment(target) - keys = (*SMART_ROUTING_ENV_KEYS, SMART_ROUTING_CONFIG_VERSION_ENV_VAR) + keys = (*SMART_ROUTING_ENV_KEYS, SMART_ROUTER_CONFIG_VERSION_ENV_VAR) previous = {key: target.get(key) for key in keys} target.update({key: resolved[key] for key in SMART_ROUTING_ENV_KEYS}) - target.pop(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, None) + target.pop(SMART_ROUTER_CONFIG_VERSION_ENV_VAR, None) return previous diff --git a/src/ucode/smart_routing/v2.py b/src/ucode/smart_routing/v2.py index 6d959e2c6..2db12b2fa 100644 --- a/src/ucode/smart_routing/v2.py +++ b/src/ucode/smart_routing/v2.py @@ -31,7 +31,7 @@ ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, LOOPBACK_HOST, - SMART_ROUTING_CONFIG_VERSION_ENV_VAR, + SMART_ROUTER_CONFIG_VERSION_ENV_VAR, SMART_ROUTING_ENV_KEYS, ) from ucode.custom_oauth import custom_oauth_cli_enabled, get_custom_client_token @@ -214,9 +214,9 @@ def disable_smart_routing( """Temporarily remove the smart-routing env vars and return their prior values.""" target = os.environ if env is None else env previous = {var: target.pop(var, None) for var in SMART_ROUTING_ENV_KEYS} - if SMART_ROUTING_CONFIG_VERSION_ENV_VAR in target: - previous[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] = target.pop( - SMART_ROUTING_CONFIG_VERSION_ENV_VAR + if SMART_ROUTER_CONFIG_VERSION_ENV_VAR in target: + previous[SMART_ROUTER_CONFIG_VERSION_ENV_VAR] = target.pop( + SMART_ROUTER_CONFIG_VERSION_ENV_VAR ) return previous diff --git a/tests/README.md b/tests/README.md index 94a6f5216..3597977e2 100644 --- a/tests/README.md +++ b/tests/README.md @@ -135,22 +135,24 @@ that Claude settings and Codex's shell policy carry the interpreter and session These are component checks; they do not establish native skill permission matching or PowerShell execution. -`test_smart_routing_config.py` is a 768-case Cartesian component oracle: all three legacy -routing flags take unset, blank, `0`, and `1`, while the selector takes unset, blank, -whitespace-only, all three canonical presets, whitespace-padded presets, unsuffixed names, -and an unsupported version. It independently hardcodes each preset and asserts exact -`resolve_environment` and `apply_config` settings, unrelated-key preservation, valid-selector -consumption, blank-selector preservation by `apply_config`, legacy-value preservation, and -nonmutation on invalid input. This file intentionally does not cover CLI startup ordering, -import-time schema validation, snapshots/restoration, routing getters, native subcommands, -hooks, session files, or managed launches; these are not established by this grid. These are -component checks; they do not establish live agent, hook, or gateway behavior. - -The toggle integration journeys run with `ENABLE_SMART_ROUTER_ORCHESTRATOR` unset and with -`ENABLE_SMART_ROUTER_ORCHESTRATOR=1`. They require only `smart-router` by default and both -bundled skills when opted in, verify the saved session controls and native +`test_smart_routing_config.py` is a 135-case Cartesian component oracle: all three legacy +routing flags take unset, `0`, and `1`, while the selector takes unset or one of the four +supported presets, including the customer first-prompt-and-subagent mode. It independently hardcodes each preset and asserts exact +`resolve_environment` and `apply_config` settings, true-unset omission, unrelated-key and +input preservation, and valid-selector consumption. This file intentionally does not cover +blank, whitespace-padded, unsuffixed, or unsupported selectors; import-time schema validation, +CLI startup ordering, snapshots/restoration, routing getters, native subcommands, hooks, +session files, or managed launches are not established by this grid. These are component +checks; they do not establish live agent, hook, or gateway behavior. + +The toggle integration journeys intentionally run with only the three subagent-only +`SMART_ROUTER_CONFIG_VERSION` presets. They require only `smart-router` for `subagent_only_v0` +and `subagent_only_v1`, and both bundled skills for `subagent_orch_v0`; they verify the saved +session controls and native tool-result confirmation after each toggle, and explicitly request their children, including while routing is off. +The managed-fixture banner journeys separately cover the selector unset (managed default) and +`first_prompt_and_subagent_no_orch_v0` for real first-prompt tasks. `test_integration_evidence.py` checks native tool-result extraction for both agents, including collapsed-output records, and excludes user echoes and assistant claims. @@ -200,16 +202,16 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_ug_codex_headless_prompt_argument`, `test_ug_codex_headless_prompt_stdin`, `test_ug_codex_headless_prompt_after_separator` | Run Codex from a script using each prompt form | Completed turn and final answer contain the file value; exit zero; no routing | | `test_ug_opencode_headless_prompt_argument` | Run OpenCode from a script (`run --format json --auto`) with an argument prompt | Completed Read tool call; final text answer contains the file value; exit zero (non-blocking CI lane) | | `test_ug_claude_exports_trace_to_configured_table`, `test_ug_codex_exports_trace_to_configured_table` | Configure tracing, complete a headless task carrying a unique trace marker, then wait for ingestion | The configured trace table contains an agent span with the same trace-safe marker and requested model | -| `test_ug_claude_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` before and after ug's separator, without workspace policy and with routing enabled | Real file task completes; JSON `modelUsage` reports the requested model with output tokens; no routing wrapper | -| `test_ug_codex_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` / `-m VALUE` with routing enabled | Real file task completes; no routing wrapper | +| `test_ug_claude_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` before and after ug's separator under each supported routing preset, without workspace policy | Real file task completes; JSON `modelUsage` reports the requested model with output tokens; no routing wrapper | +| `test_ug_codex_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` / `-m VALUE` under each supported routing preset | Real file task completes; no routing wrapper | | `test_ug_claude_preserves_caller_settings_and_hook` | Pass a settings path containing spaces | Real SessionStart hook executes; caller file unchanged; file task completes | | `test_ug_claude_reports_unsupported_short_model_option` | Pass Claude's unsupported `-m` | Actual agent error and exit status preserved | -| `test_ug_claude_auth_help`, `test_ug_claude_mcp_help` | Request subcommand help, routing off/on | Real agent help; no routing wrapper | -| `test_ug_codex_app_help`, `test_ug_codex_app_server_help`, `test_ug_codex_exec_help`, `test_ug_codex_mcp_help` | Request subcommand help, routing off/on | Real agent help; no routing wrapper | -| `test_ug_codex_app_reports_unknown_argument` | Pass an invalid option directly to `ug codex app`, routing off/on | Real Codex parser error and status preserved | -| `test_ug_codex_app_server_client_initializes` | Connect a stdio client, direct/`--` separator, routing off/on | Actual JSON-RPC initialize response; no non-JSON stdout; no routing | -| `test_smart_routing_claude_route_subagent_hook`, `test_smart_routing_codex_route_subagent_hook` | Pipe a real PreToolUse spawn payload to the installed route-subagent hook with subagent-only routing enabled | Allow decision against the live router; requested model replaced by a routed agent definition (Claude) or bundled catalog slug (Codex) from the offered models; one audited decision matching the session and task | -| `test_smart_router_skill_toggles_claude_subagent_routing`, `test_smart_router_skill_toggles_codex_subagent_routing` | Configure, launch a real subagent-only TUI with orchestration unset or opted in, then spawn tagged children while invoking the installed Smart Router skill to switch routing on -> off -> on in the same session | Only `smart-router` is installed by default; opt-in also installs `smart-router-orchestrator`; all three native children complete; only routing-enabled phases show the subagent banner and produce a live routing decision correlated with the child; no first-prompt routing wrapper; normal exit | +| `test_ug_claude_auth_help`, `test_ug_claude_mcp_help` | Request subcommand help with the selector unset and with each supported preset | Real agent help; no routing wrapper | +| `test_ug_codex_app_help`, `test_ug_codex_app_server_help`, `test_ug_codex_exec_help`, `test_ug_codex_mcp_help` | Request subcommand help with the selector unset and with each supported preset | Real agent help; no routing wrapper | +| `test_ug_codex_app_reports_unknown_argument` | Pass an invalid option directly to `ug codex app` with the selector unset and with each supported preset | Real Codex parser error and status preserved | +| `test_ug_codex_app_server_client_initializes` | Connect a stdio client, direct/`--` separator, with the selector unset and with each supported preset | Actual JSON-RPC initialize response; no non-JSON stdout; no routing | +| `test_smart_routing_claude_route_subagent_hook`, `test_smart_routing_codex_route_subagent_hook` | Pipe a real PreToolUse spawn payload to the installed route-subagent hook under each supported routing preset | Allow decision against the live router; requested model replaced by a routed agent definition (Claude) or bundled catalog slug (Codex) from the offered models; one audited decision matching the session and task | +| `test_smart_router_skill_toggles_claude_subagent_routing`, `test_smart_router_skill_toggles_codex_subagent_routing` | Configure, launch a real subagent-only TUI under the three subagent-only presets, then spawn tagged children while invoking the installed Smart Router skill to switch routing on -> off -> on in the same session | `subagent_only_v0` and `subagent_only_v1` install `smart-router`; `subagent_orch_v0` also installs `smart-router-orchestrator`; all three native children complete; only routing-enabled phases show the subagent banner and produce a live routing decision correlated with the child; no first-prompt routing wrapper; the customer full-mode preset is intentionally outside this journey; normal exit | | `test_ug_configure_claude_repeat_and_revert`, `test_ug_configure_codex_repeat_and_revert` | Configure twice over user settings; complete a task; revert twice | Settings preserved; no bearer in ug state; generated config removed; status unconfigured | | `test_ug_configure_claude_cleans_stale_skills_mcp_on_workspace_switch` | Configure the first workspace, register its skills MCP, switch to a second real workspace, and use Claude | Old registration removed from Claude and the new workspace state; old workspace bucket preserved; repeat configure stays clean; real file task completes on the second workspace | | `test_ug_configure_claude_rejects_invalid_credentials`, `test_ug_configure_codex_rejects_invalid_credentials` | Configure with a rejected bearer against the real workspace | Authentication failure; no successful saved setup | @@ -219,7 +221,7 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_case_03_*`, `test_case_05_*` | Pass a provider or model-location override to managed Claude after configure and from fresh state | ug rejects the override before Claude starts and preserves agent-owned state | | `test_case_02_*` | Launch managed Codex after configure and from fresh state | The scoped and stable catalogs, ug-launched app server, and fresh bare app server match the independently fetched admin MPS model IDs. The configured case uses real `ug revert` to remove ug's shared pointer and stable file while preserving a user setting | | `test_case_04_*`, `test_case_06_*` | Pass a provider or model-location override to managed Codex after configure and from fresh state | ug rejects the override before Codex starts and preserves agent-owned state | -| `test_ug_configure_managed_codex_catalog_fallback` | Configure from an injected managed response containing a GPT model absent from Codex's bundled catalog | Actionable metadata warning; conservative catalog entry for the unknown model; real Codex prompt on the valid default model | +| `test_ug_configure_managed_codex_catalog_fallback` | Configure from an injected managed response containing a GPT model absent from Codex's bundled catalog, then inspect the picker with the selector unset and with each supported preset | Actionable metadata warning; conservative catalog entry for the unknown model; real Codex picker lists the custom catalog model | | `test_managed_fixture_codex_http_headers_in_managed_file` | Interactive PTY configure with injected managed `http_headers` for Codex | The specified header (`x-databricks-workspace`) lands in `model_providers.Databricks.http_headers` in `/etc/codex/managed_config.toml` with the exact admin value | | `test_managed_claude_mps_defaults_accompany_discovery`, `test_managed_claude_parent_schema_defaults_accompany_discovery` | Configure from a stubbed config and launch Claude with MPS discovery (`main.default.ci_e2e_anthropic_mps`) and with `system.ai` Unity Catalog discovery, respectively, both on the managed workspace | Both generated settings files retain every admin-authored default alongside the source header and every independently fetched catalog model with its label; MPS pickers keep family shortcut rows separate from catalog entries; only UC Opus/Sonnet family ids gain `[1m]` | | `test_unmanaged_claude_preserves_preexisting_family_defaults` | Seed Claude's OS-managed family defaults, then configure against one real workspace verified to have no managed config | Every pre-existing Claude family default remains unchanged in the OS-managed settings file | @@ -230,14 +232,14 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_ug_and_ucode_auth_helpers_emit_only_the_supplied_bearer` | Run both auth helper commands with the public bearer override, with and without forced refresh | Exact token-only stdout, no warnings or ANSI escapes; no workspace authentication or saved state | | `test_ug_and_ucode_web_search_helpers_preserve_mcp_stdio` | Initialize and list tools through both web-search helper commands | Exactly the MCP JSON-RPC responses; no text/ANSI contamination; existing server/tool identities preserved; no model request | -With Claude and Codex selected there are **64 live cases** (12 marked TUI cases), +With Claude and Codex selected there are **122 live cases** (14 marked TUI cases), **1 two-workspace case** (marker `workspace_switch`), -**33 managed-fixture cases** (marker `managed_fixture`, with only +**42 managed-fixture cases** (marker `managed_fixture`, with only the CodingAgentConfig input injected from a JSON file in `fixtures/managed_config/`), and **7 installation checks**. The 14 retained numbered scenarios comprise **24 explicit journeys**: 12 managed configured/fresh executions and 12 unmanaged executions. The remaining managed-fixture cases cover focused model, MCP, skills, cache-TTL, and lifecycle shapes, including two Claude defaults cases. Parametrization varies -argument spelling or routing mode, never hides the agent/provider in the test name. Duplicate boot-only cases +argument spelling or external routing selector, never hides the agent/provider in the test name. Duplicate boot-only cases are incorporated into the Databricks configuration TUI journeys. Generated-file cleanup and strict app-server stdout assertions remain enforced. Unmanaged discovery Cases 7–14 configure, list models, or open the picker without @@ -294,7 +296,7 @@ dependency graph to reproduce a user's combination. Every relevant same-reposito PR and push to `main` runs both smoke and the full CUJ suite. Smoke covers the Databricks Hosted configure/TUI, custom OAuth CLI TUI, and headless argument journeys for both agents, in two parallel jobs. After smoke finishes, the full -suite runs all 64 live cases across two parallel agent jobs: one Claude VM and one +suite runs all 122 live cases across two parallel agent jobs: one Claude VM and one Codex VM, each running its configure, headless, and commands/lifecycle cases serially. Each agent is installed once for the full suite, and no two full jobs for the same agent overlap within a run. diff --git a/tests/conftest.py b/tests/conftest.py index c16ab3801..244b5bf2b 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -75,7 +75,7 @@ def reject_privileged_write(path, _desired_text): monkeypatch.delenv("ENABLE_SMART_ROUTING_V2", raising=False) monkeypatch.delenv("ENABLE_SMART_ROUTING_SUBAGENT_ONLY", raising=False) monkeypatch.delenv("ENABLE_SMART_ROUTER_ORCHESTRATOR", raising=False) - monkeypatch.delenv("SMART_ROUTING_CONFIG_VERSION", raising=False) + monkeypatch.delenv("SMART_ROUTER_CONFIG_VERSION", raising=False) monkeypatch.delenv("SMART_ROUTER_NAME", raising=False) # On Windows, resolve_command swaps a bare program name for whatever `shutil.which` # finds on the developer's PATH (e.g. a real `codex.CMD`). Rebind only the compatibility diff --git a/tests/integration/README.md b/tests/integration/README.md index 59c5f8808..e4e81f982 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -289,10 +289,11 @@ PATH conflicts for the Smart Router skill have subprocess/component coverage in `ug` first in PATH. The live journeys above do not inject a second installation or establish PowerShell command execution. -The toggle journeys run with `ENABLE_SMART_ROUTER_ORCHESTRATOR` unset and with -`ENABLE_SMART_ROUTER_ORCHESTRATOR=1`. They require only `smart-router` by default and both -`smart-router-orchestrator` and `smart-router` when opted in. They verify the saved session -controls, a new CLI confirmation in the native tool-result records, and a new +The toggle journeys intentionally run once under each of the three subagent-only +`SMART_ROUTER_CONFIG_VERSION` presets. They require only `smart-router` for `subagent_only_v0` +and `subagent_only_v1`, and both `smart-router-orchestrator` and `smart-router` for +`subagent_orch_v0`. They verify the saved session controls, a new CLI confirmation in the native +tool-result records, and a new assistant answer after each skill invocation. Collapsed terminal output is allowed; the answer need not repeat the CLI's exact wording. Each following child still verifies whether a routing decision occurred. @@ -303,15 +304,15 @@ role-contract preservation, and isolation from legacy preference files lack dedicated regression coverage. Codex's native hook merging, project trust, and execution of pre-existing hooks are not exercised by this integration suite. -The unit/component `../test_smart_routing_config.py` is a 768-case Cartesian oracle over all -three legacy routing flags and twelve selector forms: unset, blank, whitespace-only, the three -canonical presets, whitespace-padded presets, unsuffixed names, and an unsupported version. It -independently hardcodes preset values and asserts exact `resolve_environment` and `apply_config` -settings, unrelated-key preservation, valid-selector consumption, blank-selector preservation by -`apply_config`, legacy-value preservation, and nonmutation on invalid input. This file no longer -asserts CLI startup ordering, import-time schema validation, snapshots/restoration, routing -getters, native subcommands, hooks, session files, or managed launches; it does not claim live -agent, hook, or gateway coverage. +The unit/component `../test_smart_routing_config.py` is a 135-case Cartesian oracle over all +three legacy routing flags (`None`, `0`, `1`) and five selector forms (`None` plus the four +supported presets, including the customer first-prompt-and-subagent mode). It independently hardcodes preset values and asserts exact +`resolve_environment` and `apply_config` settings, true-unset omission, unrelated-key and +input preservation, and valid-selector consumption. This file no longer covers blank, +whitespace-padded, unsuffixed, or unsupported selectors; import-time schema validation, CLI +startup ordering, snapshots/restoration, routing getters, native subcommands, hooks, session +files, or managed launches are not established here. It does not claim live agent, hook, or +gateway coverage. The portable `../test_claude_windows_smart_routing.py` checks the Windows subagent-only fallback without Unix imports. Native Windows TUI and hook execution @@ -394,10 +395,10 @@ startup banners and footer text cannot satisfy discovery assertions. Cases 7–1 they only configure, list models, and open/close the picker. Other live CUJs perform real model tasks. -There are **64 live cases** (including 12 marked TUI journeys) and **7 installation +There are **122 live cases** (including 14 marked TUI journeys) and **7 installation checks** with Claude and Codex; selecting OpenCode adds one live headless case. One **`workspace_switch` case** uses two real workspaces and checks skills MCP cleanup and a completed -Claude task. A further **33 `managed_fixture` cases** (two of them also `live`) run on the +Claude task. A further **42 `managed_fixture` cases** (six of them also `live`) run on the managed workspace with a checked-in JSON CodingAgentConfig from `tests/fixtures/managed_config/` injected through `UCODE_MANAGED_CONFIG_STUB`; there are no cases that read a published config. Twelve explicit configured/fresh Claude and Codex discovery and source-override journeys use the @@ -557,13 +558,13 @@ each test; only explicit-model scenarios choose and record a discovered Every same-repository PR and push to `main` runs **Smoke journeys**, followed by **Full journeys** even if smoke fails. Smoke runs the Hosted configure/TUI, headless argument, and custom OAuth CLI TUI journeys for each agent (six cases, -two agent jobs). Full runs all 64 live cases, including those smoke cases, in two +two agent jobs). Full runs all 122 live cases, including those smoke cases, in two disjoint agent lanes: | Agent lane | Marker | Cases | | --- | --- | --- | -| Claude | `live and claude` | 30 | -| Codex | `live and codex` | 34 | +| Claude | `live and claude` | 53 | +| Codex | `live and codex` | 69 | A non-blocking **OpenCode** job (`live and opencode`, one case) runs alongside them with `continue-on-error` and is not part of the required `cujs` gate until it is stable. @@ -636,15 +637,16 @@ Codex state comparisons exclude `.codex/tmp/arg0`, the disposable executable lin recreated by version checks, while continuing to compare persistent agent files. In addition, `test_ug_configure_managed_codex_catalog_fallback` injects the intentionally nonexistent `system.ai.gpt-99`, keeping it out of the real workspace while launching Codex through that -workspace on the valid default model `system.ai.gpt-5-6-sol`. With smart routing enabled, it opens -the real Codex `/models` picker and requires that injected custom-catalog model to be listed. The -same picker assertion also runs with smart routing disabled to cover both launch paths. +workspace on the valid default model `system.ai.gpt-5-6-sol`. With the selector unset and with +each supported `SMART_ROUTER_CONFIG_VERSION` preset, it opens the real Codex `/models` picker +and requires that injected custom-catalog model to be listed. The fixture itself has no managed +smart-routing setting, so the unset case remains the unconfigured baseline. The smart-routing banner journeys inject static Claude and Codex model lists with `smart_routing` enabled in the agent config, run `ug configure`, then launch the real TUI and -submit one small file task. Each asserts the "Using Unity Gateway Smart Router." banner naming -the selected model appears in the TUI, the routed answer completes the file task, and the -session exits normally. +submit one small file task. Each runs once with the selector unset (managed default) and once +with `first_prompt_and_subagent_no_orch_v0`, asserting the "Using Unity Gateway Smart Router." +banner naming the selected model, a routed answer completing the file task, and normal exit. That workspace authenticates as a service principal, so CI mints a short-lived token per run from these same-repository secrets rather than storing a long-lived bearer: @@ -867,7 +869,7 @@ uv run --no-project --python 3.12 python scripts/run_integration.py \ unset DATABRICKS_BEARER ``` -This runs all 64 live cases. For the seven installation checks, run the same +This runs all 122 live cases. For the seven installation checks, run the same runner/version/index arguments with `--installation-only` and omit `-- -m live`; no bearer or workspace is needed. Results remain under `.integration-runs/`. Each invocation needs a new output directory; an existing one is rejected. diff --git a/tests/integration/test_ug_claude_commands.py b/tests/integration/test_ug_claude_commands.py index 701c624ab..099bc51eb 100644 --- a/tests/integration/test_ug_claude_commands.py +++ b/tests/integration/test_ug_claude_commands.py @@ -5,9 +5,25 @@ pytestmark = [pytest.mark.live, pytest.mark.claude] -@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_ug_claude_auth_help(live_session, workspace, routing): - """Scenario: configure claude, then ask ug for auth subcommand help. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + None, + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], + ids=[ + "unconfigured", + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], +) +def test_ug_claude_auth_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION): + """Scenario: configure claude, then ask ug for auth help under each selector. Expected: the real claude help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -23,16 +39,33 @@ def test_ug_claude_auth_help(live_session, workspace, routing): "--skip-upgrade", "--disable-databricks-ai-tools", ) - session.env["ENABLE_SMART_ROUTING_V2"] = routing + if SMART_ROUTER_CONFIG_VERSION is not None: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION expected = session.run("auth", "--help", binary="claude").stdout.strip() actual = session.run("claude", "--", "auth", "--help").stdout assert expected and expected in actual, actual session.assert_not_routed() -@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_ug_claude_mcp_help(live_session, workspace, routing): - """Scenario: configure claude, then ask ug for mcp subcommand help. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + None, + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], + ids=[ + "unconfigured", + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], +) +def test_ug_claude_mcp_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION): + """Scenario: configure claude, then ask ug for mcp help under each selector. Expected: the real claude help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -48,7 +81,8 @@ def test_ug_claude_mcp_help(live_session, workspace, routing): "--skip-upgrade", "--disable-databricks-ai-tools", ) - session.env["ENABLE_SMART_ROUTING_V2"] = routing + if SMART_ROUTER_CONFIG_VERSION is not None: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION expected = session.run("mcp", "--help", binary="claude").stdout.strip() actual = session.run("claude", "--", "mcp", "--help").stdout assert expected and expected in actual, actual diff --git a/tests/integration/test_ug_claude_headless.py b/tests/integration/test_ug_claude_headless.py index 4a41204de..bd83d3b00 100644 --- a/tests/integration/test_ug_claude_headless.py +++ b/tests/integration/test_ug_claude_headless.py @@ -116,12 +116,27 @@ def test_ug_claude_headless_prompt_after_separator(live_session, workspace): @pytest.mark.parametrize("model_form", ["separate", "equals"]) @pytest.mark.parametrize("model_owner", ["ug", "claude"]) +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], + ids=[ + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], +) def test_ug_claude_headless_explicit_model_bypasses_routing( - live_session, workspace, model_form, model_owner + live_session, workspace, model_form, model_owner, SMART_ROUTER_CONFIG_VERSION ): - """Scenario: choose a model before/after ug's separator with no workspace policy. + """Scenario: choose a model before/after ug's separator under each supported selector. - Expected: with smart routing enabled, the real file task completes on the + Expected: under each supported routing preset, the real file task completes on the requested model, confirmed by JSON modelUsage, without a routing wrapper. """ session = live_session @@ -138,7 +153,7 @@ def test_ug_claude_headless_explicit_model_bypasses_routing( ) model = session.model_for_explicit_case("claude") model_args = ["--model", model] if model_form == "separate" else [f"--model={model}"] - session.env["ENABLE_SMART_ROUTING_V2"] = "1" + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION result = session.run( "claude", *(model_args if model_owner == "ug" else []), diff --git a/tests/integration/test_ug_codex_app_server.py b/tests/integration/test_ug_codex_app_server.py index 9845dd190..0b791b2ea 100644 --- a/tests/integration/test_ug_codex_app_server.py +++ b/tests/integration/test_ug_codex_app_server.py @@ -6,9 +6,27 @@ @pytest.mark.parametrize("separator", [False, True], ids=["direct", "launcher-separator"]) -@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_ug_codex_app_server_client_initializes(live_session, workspace, separator, routing): - """Scenario: configure Codex and connect a real stdio client to ug codex app-server. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + None, + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], + ids=[ + "unconfigured", + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], +) +def test_ug_codex_app_server_client_initializes( + live_session, workspace, separator, SMART_ROUTER_CONFIG_VERSION +): + """Scenario: configure Codex and connect a real stdio client under each selector. Expected: initialize returns a valid JSON-RPC result, diagnostics stay off the protocol stream, and this utility command never starts smart routing. @@ -24,7 +42,8 @@ def test_ug_codex_app_server_client_initializes(live_session, workspace, separat "--skip-upgrade", "--disable-databricks-ai-tools", ) - session.env["ENABLE_SMART_ROUTING_V2"] = routing + if SMART_ROUTER_CONFIG_VERSION is not None: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION args = ["app-server", "--listen", "stdio://"] if separator: diff --git a/tests/integration/test_ug_codex_commands.py b/tests/integration/test_ug_codex_commands.py index fe382fba5..10adff908 100644 --- a/tests/integration/test_ug_codex_commands.py +++ b/tests/integration/test_ug_codex_commands.py @@ -5,9 +5,25 @@ pytestmark = [pytest.mark.live, pytest.mark.codex] -@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_ug_codex_app_help(live_session, workspace, routing): - """Scenario: configure codex, then ask ug for app subcommand help. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + None, + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], + ids=[ + "unconfigured", + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], +) +def test_ug_codex_app_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION): + """Scenario: configure codex, then ask ug for app subcommand help under each selector. Expected: the real codex help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -23,16 +39,33 @@ def test_ug_codex_app_help(live_session, workspace, routing): "--skip-upgrade", "--disable-databricks-ai-tools", ) - session.env["ENABLE_SMART_ROUTING_V2"] = routing + if SMART_ROUTER_CONFIG_VERSION is not None: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION expected = session.run("app", "--help", binary="codex").stdout.strip() actual = session.run("codex", "--", "app", "--help").stdout assert expected and expected in actual, actual session.assert_not_routed() -@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_ug_codex_app_server_help(live_session, workspace, routing): - """Scenario: configure codex, then ask ug for app-server subcommand help. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + None, + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], + ids=[ + "unconfigured", + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], +) +def test_ug_codex_app_server_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION): + """Scenario: configure codex, then ask ug for app-server help under each selector. Expected: the real codex help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -48,16 +81,33 @@ def test_ug_codex_app_server_help(live_session, workspace, routing): "--skip-upgrade", "--disable-databricks-ai-tools", ) - session.env["ENABLE_SMART_ROUTING_V2"] = routing + if SMART_ROUTER_CONFIG_VERSION is not None: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION expected = session.run("app-server", "--help", binary="codex").stdout.strip() actual = session.run("codex", "--", "app-server", "--help").stdout assert expected and expected in actual, actual session.assert_not_routed() -@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_ug_codex_exec_help(live_session, workspace, routing): - """Scenario: configure codex, then ask ug for exec subcommand help. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + None, + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], + ids=[ + "unconfigured", + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], +) +def test_ug_codex_exec_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION): + """Scenario: configure codex, then ask ug for exec help under each selector. Expected: the real codex help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -73,16 +123,33 @@ def test_ug_codex_exec_help(live_session, workspace, routing): "--skip-upgrade", "--disable-databricks-ai-tools", ) - session.env["ENABLE_SMART_ROUTING_V2"] = routing + if SMART_ROUTER_CONFIG_VERSION is not None: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION expected = session.run("exec", "--help", binary="codex").stdout.strip() actual = session.run("codex", "--", "exec", "--help").stdout assert expected and expected in actual, actual session.assert_not_routed() -@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_ug_codex_mcp_help(live_session, workspace, routing): - """Scenario: configure codex, then ask ug for mcp subcommand help. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + None, + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], + ids=[ + "unconfigured", + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], +) +def test_ug_codex_mcp_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION): + """Scenario: configure codex, then ask ug for mcp help under each selector. Expected: the real codex help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -98,16 +165,35 @@ def test_ug_codex_mcp_help(live_session, workspace, routing): "--skip-upgrade", "--disable-databricks-ai-tools", ) - session.env["ENABLE_SMART_ROUTING_V2"] = routing + if SMART_ROUTER_CONFIG_VERSION is not None: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION expected = session.run("mcp", "--help", binary="codex").stdout.strip() actual = session.run("codex", "--", "mcp", "--help").stdout assert expected and expected in actual, actual session.assert_not_routed() -@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_ug_codex_app_reports_unknown_argument(live_session, workspace, routing): - """Scenario: pass an unknown option directly to ug codex app. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + None, + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], + ids=[ + "unconfigured", + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], +) +def test_ug_codex_app_reports_unknown_argument( + live_session, workspace, SMART_ROUTER_CONFIG_VERSION +): + """Scenario: pass an unknown option directly to ug codex app under each selector. Expected: the actual Codex parser's error and exit status are preserved, without opening a desktop application or starting routing. @@ -123,7 +209,8 @@ def test_ug_codex_app_reports_unknown_argument(live_session, workspace, routing) "--skip-upgrade", "--disable-databricks-ai-tools", ) - session.env["ENABLE_SMART_ROUTING_V2"] = routing + if SMART_ROUTER_CONFIG_VERSION is not None: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION args = ["app", "--ug-integration-unknown-option"] expected = session.run(*args, binary="codex", ok=False) actual = session.run("codex", *args, ok=False) diff --git a/tests/integration/test_ug_codex_headless.py b/tests/integration/test_ug_codex_headless.py index b58f5dcd8..2a9378337 100644 --- a/tests/integration/test_ug_codex_headless.py +++ b/tests/integration/test_ug_codex_headless.py @@ -119,8 +119,25 @@ def test_ug_codex_headless_prompt_after_separator(live_session, workspace): @pytest.mark.parametrize("model_form", ["separate", "equals", "short"]) -def test_ug_codex_headless_explicit_model_bypasses_routing(live_session, workspace, model_form): - """Scenario: choose an explicit model while global smart routing is enabled. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], + ids=[ + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], +) +def test_ug_codex_headless_explicit_model_bypasses_routing( + live_session, workspace, model_form, SMART_ROUTER_CONFIG_VERSION +): + """Scenario: choose an explicit model under each supported routing selector. Expected: the model option is accepted, the real file task completes, and no routing wrapper overrides the caller's choice. @@ -141,7 +158,7 @@ def test_ug_codex_headless_explicit_model_bypasses_routing(live_session, workspa model_args = ["--model", model] if model_form == "separate" else [f"--model={model}"] if model_form == "short": model_args = ["-m", model] - session.env["ENABLE_SMART_ROUTING_V2"] = "1" + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION result = session.run( "codex", "--", diff --git a/tests/integration/test_ug_configure_managed_models.py b/tests/integration/test_ug_configure_managed_models.py index 5dd34bb31..b4c6e2fe5 100644 --- a/tests/integration/test_ug_configure_managed_models.py +++ b/tests/integration/test_ug_configure_managed_models.py @@ -171,13 +171,31 @@ def test_managed_fixture_claude_model_picker_reflects_the_config(live_session, w @pytest.mark.managed_fixture @pytest.mark.codex -@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_ug_configure_managed_codex_catalog_fallback(live_session, workspace, routing): - """Scenario: configure Codex from an injected model list containing an unknown GPT model. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + None, + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], + ids=[ + "unconfigured", + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], +) +def test_ug_configure_managed_codex_catalog_fallback( + live_session, workspace, SMART_ROUTER_CONFIG_VERSION +): + """Scenario: configure Codex from an injected model list under each selector. Expected: ug creates conservative fallback metadata for the unknown model, warns how to get - richer metadata, and the real Codex TUI lists that model in its /model picker both with and - without smart routing. + richer metadata, and the real Codex TUI lists that model in its /model picker for the + unconfigured case and each supported routing selector. """ session = live_session use_managed_config_fixture(session, "codex_catalog_fallback") @@ -199,9 +217,13 @@ def test_ug_configure_managed_codex_catalog_fallback(live_session, workspace, ro assert fallback.get("context_window") == 32768, fallback assert fallback.get("default_reasoning_level") == "none", fallback - session.env["ENABLE_SMART_ROUTING_V2"] = routing + if SMART_ROUTER_CONFIG_VERSION is not None: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION with AgentTerminal( - session, "codex", [str(session.binary), "codex"], f"managed-fallback-routing-{routing}" + session, + "codex", + [str(session.binary), "codex"], + f"managed-fallback-routing-{SMART_ROUTER_CONFIG_VERSION or 'unconfigured'}", ) as tui: tui.boot() tui.submit("/model") @@ -217,22 +239,35 @@ def test_ug_configure_managed_codex_catalog_fallback(live_session, workspace, ro @pytest.mark.managed_fixture @pytest.mark.claude -def test_managed_fixture_claude_smart_routing_banner(live_session, workspace): - """Scenario: an admin config lists Claude models and enables smart routing for Claude. - - Expected: `ug configure` applies the config without the personal agent selector, and - `ug claude` routes the first real prompt: the TUI shows the Unity Gateway Smart Router - banner naming the selected model, the routed answer completes the file task, and the - session exits normally. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [None, "first_prompt_and_subagent_no_orch_v0"], + ids=["managed-default", "first_prompt_and_subagent_no_orch_v0"], +) +def test_managed_fixture_claude_smart_routing_banner( + live_session, workspace, SMART_ROUTER_CONFIG_VERSION +): + """Scenario: an admin config lists Claude models and launch uses its default or the + customer first-prompt-and-subagent selector. + + Expected: `ug configure` applies the config without the personal agent selector, and both + launch modes route the first real prompt: the TUI shows the Unity Gateway Smart Router + banner naming the selected model, the routed answer completes the file task, and the session + exits normally. """ session = live_session task = FileTask(session) use_managed_config_fixture(session, "claude_smart_routing") result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) assert "Select coding agents to configure:" not in result.stdout, result.stdout + if SMART_ROUTER_CONFIG_VERSION is not None: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION with AgentTerminal( - session, "claude", [str(session.binary), "claude"], "managed-smart-routing" + session, + "claude", + [str(session.binary), "claude"], + f"managed-smart-routing-{SMART_ROUTER_CONFIG_VERSION or 'managed-default'}", ) as tui: tui.boot() tui.submit(task.prompt) @@ -249,22 +284,35 @@ def test_managed_fixture_claude_smart_routing_banner(live_session, workspace): @pytest.mark.managed_fixture @pytest.mark.codex -def test_managed_fixture_codex_smart_routing_banner(live_session, workspace): - """Scenario: an admin config lists Codex models and enables smart routing for Codex. - - Expected: `ug configure` applies the config without the personal agent selector, and - `ug codex` routes the first real prompt: the TUI shows the Unity Gateway Smart Router - banner naming the selected model, the routed answer completes the file task, and the - session exits normally. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [None, "first_prompt_and_subagent_no_orch_v0"], + ids=["managed-default", "first_prompt_and_subagent_no_orch_v0"], +) +def test_managed_fixture_codex_smart_routing_banner( + live_session, workspace, SMART_ROUTER_CONFIG_VERSION +): + """Scenario: an admin config lists Codex models and launch uses its default or the + customer first-prompt-and-subagent selector. + + Expected: `ug configure` applies the config without the personal agent selector, and both + launch modes route the first real prompt: the TUI shows the Unity Gateway Smart Router + banner naming the selected model, the routed answer completes the file task, and the session + exits normally. """ session = live_session task = FileTask(session) use_managed_config_fixture(session, "codex_smart_routing") result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) assert "Select coding agents to configure:" not in result.stdout, result.stdout + if SMART_ROUTER_CONFIG_VERSION is not None: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION with AgentTerminal( - session, "codex", [str(session.binary), "codex"], "managed-smart-routing" + session, + "codex", + [str(session.binary), "codex"], + f"managed-smart-routing-{SMART_ROUTER_CONFIG_VERSION or 'managed-default'}", ) as tui: tui.boot() tui.submit(task.prompt) diff --git a/tests/integration/test_ug_smart_routing_hooks.py b/tests/integration/test_ug_smart_routing_hooks.py index fc4c92db9..0515efae6 100644 --- a/tests/integration/test_ug_smart_routing_hooks.py +++ b/tests/integration/test_ug_smart_routing_hooks.py @@ -111,7 +111,9 @@ def _run_calculation(tui, session, agent: str, expression: str, expected: str, * ) -def _toggle_with_skill(tui, session, agent: str, enabled: bool) -> None: +def _toggle_with_skill( + tui, session, agent: str, enabled: bool, SMART_ROUTER_CONFIG_VERSION: str +) -> None: skill_root = session.home / SKILL_ROOTS[agent] ignored_skills = {".system"} if agent == "codex" else set() installed_skills = sorted( @@ -121,7 +123,7 @@ def _toggle_with_skill(tui, session, agent: str, enabled: bool) -> None: ) expected_skills = ( ["smart-router", "smart-router-orchestrator"] - if session.env.get("ENABLE_SMART_ROUTER_ORCHESTRATOR") == "1" + if (SMART_ROUTER_CONFIG_VERSION == "subagent_orch_v0") else ["smart-router"] ) assert installed_skills == expected_skills, installed_skills @@ -136,6 +138,7 @@ def _toggle_with_skill(tui, session, agent: str, enabled: bool) -> None: else { "ENABLE_SMART_ROUTING_V2": "0", "ENABLE_SMART_ROUTING_SUBAGENT_ONLY": "0", + "ENABLE_SMART_ROUTER_ORCHESTRATOR": "0", } ) assert json.loads(controls[0].read_text()) != expected @@ -171,8 +174,25 @@ def toggled(_screen): @pytest.mark.live @pytest.mark.claude -def test_smart_routing_claude_route_subagent_hook(live_session, workspace): - """Scenario: with subagent-only routing enabled, Claude Code fires PreToolUse for an +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], + ids=[ + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], +) +def test_smart_routing_claude_route_subagent_hook( + live_session, workspace, SMART_ROUTER_CONFIG_VERSION +): + """Scenario: each supported routing selector makes Claude Code fire PreToolUse for an Agent spawn, piping the payload to ``ug claude-router-hook route-subagent``. Expected: the hook allows the call against the real workspace router, drops the @@ -181,7 +201,7 @@ def test_smart_routing_claude_route_subagent_hook(live_session, workspace): hook contract is asserted; no agent decides to spawn. """ session = live_session - session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION payload = { "session_id": "claude-route-subagent-hook", "tool_name": "Agent", @@ -225,8 +245,25 @@ def test_smart_routing_claude_route_subagent_hook(live_session, workspace): @pytest.mark.live @pytest.mark.codex -def test_smart_routing_codex_route_subagent_hook(live_session, workspace): - """Scenario: with subagent-only routing enabled, Codex fires PreToolUse for a +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], + ids=[ + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], +) +def test_smart_routing_codex_route_subagent_hook( + live_session, workspace, SMART_ROUTER_CONFIG_VERSION +): + """Scenario: each supported routing selector makes Codex fire PreToolUse for a spawn_agent call, piping the payload to ``ug codex-router-hook route-subagent``. Expected: the hook allows the call against the real workspace router, rewrites the @@ -235,7 +272,7 @@ def test_smart_routing_codex_route_subagent_hook(live_session, workspace): hook contract is asserted; no agent decides to spawn. """ session = live_session - session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION payload = { "session_id": "codex-route-subagent-hook", "tool_name": "spawn_agent", @@ -279,29 +316,27 @@ def test_smart_routing_codex_route_subagent_hook(live_session, workspace): @pytest.mark.claude @pytest.mark.managed_fixture @pytest.mark.parametrize( - "orchestration_enabled", [False, True], ids=["routing-only", "orchestration"] + "SMART_ROUTER_CONFIG_VERSION", + ["subagent_only_v0", "subagent_only_v1", "subagent_orch_v0"], + ids=["subagent_only_v0", "subagent_only_v1", "subagent_orch_v0"], ) def test_smart_router_skill_toggles_claude_subagent_routing( - live_session, workspace, tmp_path, orchestration_enabled + live_session, workspace, tmp_path, SMART_ROUTER_CONFIG_VERSION ): - """Scenario: launch Claude with subagent routing enabled and orchestration unset - or opted in through ENABLE_SMART_ROUTER_ORCHESTRATOR=1, spawn a child, invoke the - installed Smart Router skill to turn routing off, spawn another child, turn routing + """Scenario: launch Claude with each supported subagent selector, spawn a child, invoke + the installed Smart Router skill to turn routing off, spawn another child, turn routing back on through the skill, and spawn a third child in the same real TUI session. - Expected: only Smart Router is installed by default; opting in also installs smart-router-orchestrator. - Each invocation records the CLI confirmation in the native transcript and changes the saved - routing controls, even with collapsed terminal output; all three uniquely tagged - calculations complete in native child sessions; only the first and third show the - subagent-routing banner and produce live gateway decisions correlated with those children. - No first-prompt routing wrapper starts. + Expected: subagent_only_v0 and subagent_only_v1 install Smart Router, while + subagent_orch_v0 also installs Smart Router Orchestrator. Each invocation records the CLI + confirmation in the native transcript and changes the saved routing controls, even with + collapsed terminal output; all three uniquely tagged calculations complete in native child + sessions; only the first and third show the subagent-routing banner and produce live gateway + decisions correlated with those children. No first-prompt routing wrapper starts. """ session = live_session session.env["TMPDIR"] = str(tmp_path) - session.env["ENABLE_SMART_ROUTING_V2"] = "1" - session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" - if orchestration_enabled: - session.env["ENABLE_SMART_ROUTER_ORCHESTRATOR"] = "1" + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION use_managed_config_fixture(session, "claude_smart_routing") session.run( "configure", @@ -316,9 +351,21 @@ def test_smart_router_skill_toggles_claude_subagent_routing( ) as tui: tui.boot() _run_calculation(tui, session, "claude", "1+1", "2", routed=True) - _toggle_with_skill(tui, session, "claude", enabled=False) + _toggle_with_skill( + tui, + session, + "claude", + enabled=False, + SMART_ROUTER_CONFIG_VERSION=SMART_ROUTER_CONFIG_VERSION, + ) _run_calculation(tui, session, "claude", "1+2", "3", routed=False) - _toggle_with_skill(tui, session, "claude", enabled=True) + _toggle_with_skill( + tui, + session, + "claude", + enabled=True, + SMART_ROUTER_CONFIG_VERSION=SMART_ROUTER_CONFIG_VERSION, + ) _run_calculation(tui, session, "claude", "2+2", "4", routed=True) tui.exit_normally() transcript = "".join(tui.output) @@ -332,29 +379,27 @@ def test_smart_router_skill_toggles_claude_subagent_routing( @pytest.mark.codex @pytest.mark.managed_fixture @pytest.mark.parametrize( - "orchestration_enabled", [False, True], ids=["routing-only", "orchestration"] + "SMART_ROUTER_CONFIG_VERSION", + ["subagent_only_v0", "subagent_only_v1", "subagent_orch_v0"], + ids=["subagent_only_v0", "subagent_only_v1", "subagent_orch_v0"], ) def test_smart_router_skill_toggles_codex_subagent_routing( - live_session, workspace, tmp_path, orchestration_enabled + live_session, workspace, tmp_path, SMART_ROUTER_CONFIG_VERSION ): - """Scenario: launch Codex with subagent routing enabled and orchestration unset - or opted in through ENABLE_SMART_ROUTER_ORCHESTRATOR=1, spawn a child, invoke the - installed Smart Router skill to turn routing off, spawn another child, turn routing + """Scenario: launch Codex with each supported subagent selector, spawn a child, invoke + the installed Smart Router skill to turn routing off, spawn another child, turn routing back on through the skill, and spawn a third child in the same real TUI session. - Expected: only Smart Router is installed by default; opting in also installs smart-router-orchestrator. - Each invocation records the CLI confirmation in the native transcript and changes the saved - routing controls, even with collapsed terminal output; all three uniquely tagged - calculations complete in native child sessions; only the first and third show the - subagent-routing banner and produce live gateway decisions correlated with those children. - No first-prompt interposer starts. + Expected: subagent_only_v0 and subagent_only_v1 install Smart Router, while + subagent_orch_v0 also installs Smart Router Orchestrator. Each invocation records the CLI + confirmation in the native transcript and changes the saved routing controls, even with + collapsed terminal output; all three uniquely tagged calculations complete in native child + sessions; only the first and third show the subagent-routing banner and produce live gateway + decisions correlated with those children. No first-prompt interposer starts. """ session = live_session session.env["TMPDIR"] = str(tmp_path) - session.env["ENABLE_SMART_ROUTING_V2"] = "1" - session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" - if orchestration_enabled: - session.env["ENABLE_SMART_ROUTER_ORCHESTRATOR"] = "1" + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION use_managed_config_fixture(session, "codex_smart_routing") session.run( "configure", @@ -369,9 +414,21 @@ def test_smart_router_skill_toggles_codex_subagent_routing( ) as tui: tui.boot() _run_calculation(tui, session, "codex", "1+1", "2", routed=True) - _toggle_with_skill(tui, session, "codex", enabled=False) + _toggle_with_skill( + tui, + session, + "codex", + enabled=False, + SMART_ROUTER_CONFIG_VERSION=SMART_ROUTER_CONFIG_VERSION, + ) _run_calculation(tui, session, "codex", "1+2", "3", routed=False) - _toggle_with_skill(tui, session, "codex", enabled=True) + _toggle_with_skill( + tui, + session, + "codex", + enabled=True, + SMART_ROUTER_CONFIG_VERSION=SMART_ROUTER_CONFIG_VERSION, + ) _run_calculation(tui, session, "codex", "2+2", "4", routed=True) tui.exit_normally() transcript = "".join(tui.output) diff --git a/tests/test_smart_routing_config.py b/tests/test_smart_routing_config.py index 2a2c4462f..d7a15e941 100644 --- a/tests/test_smart_routing_config.py +++ b/tests/test_smart_routing_config.py @@ -1,4 +1,4 @@ -"""Exhaustive component coverage for versioned smart-routing configuration.""" +"""Component coverage for supported versioned smart-routing configuration.""" from __future__ import annotations @@ -8,11 +8,16 @@ ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, - SMART_ROUTING_CONFIG_VERSION_ENV_VAR, + SMART_ROUTER_CONFIG_VERSION_ENV_VAR, ) from ucode.smart_routing import config _EXPECTED_PRESETS = { + "first_prompt_and_subagent_no_orch_v0": { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + }, "subagent_only_v0": { ENABLE_SMART_ROUTING_ENV_VAR: "0", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", @@ -31,59 +36,43 @@ } -@pytest.mark.parametrize("v2_value", [None, "", "0", "1"]) -@pytest.mark.parametrize("subagent_value", [None, "", "0", "1"]) -@pytest.mark.parametrize("orchestrator_value", [None, "", "0", "1"]) +@pytest.mark.parametrize("ENABLE_SMART_ROUTING_V2", [None, "0", "1"]) +@pytest.mark.parametrize("ENABLE_SMART_ROUTING_SUBAGENT_ONLY", [None, "0", "1"]) +@pytest.mark.parametrize("ENABLE_SMART_ROUTER_ORCHESTRATOR", [None, "0", "1"]) @pytest.mark.parametrize( - "config_version", + "SMART_ROUTER_CONFIG_VERSION", [ None, - "", - " \t", + "first_prompt_and_subagent_no_orch_v0", "subagent_only_v0", "subagent_only_v1", "subagent_orch_v0", - " subagent_only_v0 ", - " subagent_only_v1 ", - " subagent_orch_v0 ", - "subagent_only", - "subagent_orch", - "future_mode", ], ) def test_smart_routing_config_cartesian_grid( - v2_value, subagent_value, orchestrator_value, config_version + ENABLE_SMART_ROUTING_V2, + ENABLE_SMART_ROUTING_SUBAGENT_ONLY, + ENABLE_SMART_ROUTER_ORCHESTRATOR, + SMART_ROUTER_CONFIG_VERSION, ): source = { environment_key: value for environment_key, value in ( - (ENABLE_SMART_ROUTING_ENV_VAR, v2_value), - (ENABLE_SUBAGENT_ROUTING_ENV_VAR, subagent_value), - (ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, orchestrator_value), + (ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SMART_ROUTING_V2), + (ENABLE_SUBAGENT_ROUTING_ENV_VAR, ENABLE_SMART_ROUTING_SUBAGENT_ONLY), + (ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, ENABLE_SMART_ROUTER_ORCHESTRATOR), ) if value is not None } source["UNRELATED_SETTING"] = "preserved" - if config_version is not None: - source[SMART_ROUTING_CONFIG_VERSION_ENV_VAR] = config_version + if SMART_ROUTER_CONFIG_VERSION is not None: + source[SMART_ROUTER_CONFIG_VERSION_ENV_VAR] = SMART_ROUTER_CONFIG_VERSION original = source.copy() - normalized_version = (config_version or "").strip() - - if normalized_version and normalized_version not in _EXPECTED_PRESETS: - with pytest.raises(RuntimeError): - config.resolve_environment(source) - assert source == original - - applied = source.copy() - with pytest.raises(RuntimeError): - config.apply_config(applied) - assert applied == original - return expected = original.copy() - expected.pop(SMART_ROUTING_CONFIG_VERSION_ENV_VAR, None) - if normalized_version: - expected.update(_EXPECTED_PRESETS[normalized_version]) + expected.pop(SMART_ROUTER_CONFIG_VERSION_ENV_VAR, None) + if SMART_ROUTER_CONFIG_VERSION is not None: + expected.update(_EXPECTED_PRESETS[SMART_ROUTER_CONFIG_VERSION]) resolved = config.resolve_environment(source) @@ -93,5 +82,4 @@ def test_smart_routing_config_cartesian_grid( applied = source.copy() config.apply_config(applied) - expected_applied = expected if normalized_version else original - assert applied == expected_applied + assert applied == expected From 9bc917d1521c4d66e98f00b9b1e7626c0189f9c3 Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Thu, 8 Oct 2026 22:03:40 +0000 Subject: [PATCH 11/30] Limit smart router preset coverage to routing tests --- tests/README.md | 19 +-- tests/integration/README.md | 25 ++-- tests/integration/test_ug_claude_commands.py | 46 +------ tests/integration/test_ug_codex_app_server.py | 26 +--- tests/integration/test_ug_codex_commands.py | 117 ++---------------- .../test_ug_configure_managed_models.py | 30 +---- 6 files changed, 45 insertions(+), 218 deletions(-) diff --git a/tests/README.md b/tests/README.md index 3597977e2..4e66e3c9a 100644 --- a/tests/README.md +++ b/tests/README.md @@ -153,6 +153,9 @@ tool-result confirmation after each toggle, and explicitly request their childre including while routing is off. The managed-fixture banner journeys separately cover the selector unset (managed default) and `first_prompt_and_subagent_no_orch_v0` for real first-prompt tasks. +Preset parameterization is limited to routing hooks, skill toggles, explicit-model routing +bypass, and first-prompt routing. Command forwarding, app-server initialization, and catalog +fallback run without a routing selector; preset-specific behavior there is not covered. `test_integration_evidence.py` checks native tool-result extraction for both agents, including collapsed-output records, and excludes user echoes and assistant claims. @@ -206,10 +209,10 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_ug_codex_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` / `-m VALUE` under each supported routing preset | Real file task completes; no routing wrapper | | `test_ug_claude_preserves_caller_settings_and_hook` | Pass a settings path containing spaces | Real SessionStart hook executes; caller file unchanged; file task completes | | `test_ug_claude_reports_unsupported_short_model_option` | Pass Claude's unsupported `-m` | Actual agent error and exit status preserved | -| `test_ug_claude_auth_help`, `test_ug_claude_mcp_help` | Request subcommand help with the selector unset and with each supported preset | Real agent help; no routing wrapper | -| `test_ug_codex_app_help`, `test_ug_codex_app_server_help`, `test_ug_codex_exec_help`, `test_ug_codex_mcp_help` | Request subcommand help with the selector unset and with each supported preset | Real agent help; no routing wrapper | -| `test_ug_codex_app_reports_unknown_argument` | Pass an invalid option directly to `ug codex app` with the selector unset and with each supported preset | Real Codex parser error and status preserved | -| `test_ug_codex_app_server_client_initializes` | Connect a stdio client, direct/`--` separator, with the selector unset and with each supported preset | Actual JSON-RPC initialize response; no non-JSON stdout; no routing | +| `test_ug_claude_auth_help`, `test_ug_claude_mcp_help` | Request subcommand help | Real agent help; no routing wrapper | +| `test_ug_codex_app_help`, `test_ug_codex_app_server_help`, `test_ug_codex_exec_help`, `test_ug_codex_mcp_help` | Request subcommand help | Real agent help; no routing wrapper | +| `test_ug_codex_app_reports_unknown_argument` | Pass an invalid option directly to `ug codex app` | Real Codex parser error and status preserved | +| `test_ug_codex_app_server_client_initializes` | Connect a stdio client, direct/`--` separator | Actual JSON-RPC initialize response; no non-JSON stdout; no routing | | `test_smart_routing_claude_route_subagent_hook`, `test_smart_routing_codex_route_subagent_hook` | Pipe a real PreToolUse spawn payload to the installed route-subagent hook under each supported routing preset | Allow decision against the live router; requested model replaced by a routed agent definition (Claude) or bundled catalog slug (Codex) from the offered models; one audited decision matching the session and task | | `test_smart_router_skill_toggles_claude_subagent_routing`, `test_smart_router_skill_toggles_codex_subagent_routing` | Configure, launch a real subagent-only TUI under the three subagent-only presets, then spawn tagged children while invoking the installed Smart Router skill to switch routing on -> off -> on in the same session | `subagent_only_v0` and `subagent_only_v1` install `smart-router`; `subagent_orch_v0` also installs `smart-router-orchestrator`; all three native children complete; only routing-enabled phases show the subagent banner and produce a live routing decision correlated with the child; no first-prompt routing wrapper; the customer full-mode preset is intentionally outside this journey; normal exit | | `test_ug_configure_claude_repeat_and_revert`, `test_ug_configure_codex_repeat_and_revert` | Configure twice over user settings; complete a task; revert twice | Settings preserved; no bearer in ug state; generated config removed; status unconfigured | @@ -221,7 +224,7 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_case_03_*`, `test_case_05_*` | Pass a provider or model-location override to managed Claude after configure and from fresh state | ug rejects the override before Claude starts and preserves agent-owned state | | `test_case_02_*` | Launch managed Codex after configure and from fresh state | The scoped and stable catalogs, ug-launched app server, and fresh bare app server match the independently fetched admin MPS model IDs. The configured case uses real `ug revert` to remove ug's shared pointer and stable file while preserving a user setting | | `test_case_04_*`, `test_case_06_*` | Pass a provider or model-location override to managed Codex after configure and from fresh state | ug rejects the override before Codex starts and preserves agent-owned state | -| `test_ug_configure_managed_codex_catalog_fallback` | Configure from an injected managed response containing a GPT model absent from Codex's bundled catalog, then inspect the picker with the selector unset and with each supported preset | Actionable metadata warning; conservative catalog entry for the unknown model; real Codex picker lists the custom catalog model | +| `test_ug_configure_managed_codex_catalog_fallback` | Configure from an injected managed response containing a GPT model absent from Codex's bundled catalog, then inspect the picker | Actionable metadata warning; conservative catalog entry for the unknown model; real Codex picker lists the custom catalog model | | `test_managed_fixture_codex_http_headers_in_managed_file` | Interactive PTY configure with injected managed `http_headers` for Codex | The specified header (`x-databricks-workspace`) lands in `model_providers.Databricks.http_headers` in `/etc/codex/managed_config.toml` with the exact admin value | | `test_managed_claude_mps_defaults_accompany_discovery`, `test_managed_claude_parent_schema_defaults_accompany_discovery` | Configure from a stubbed config and launch Claude with MPS discovery (`main.default.ci_e2e_anthropic_mps`) and with `system.ai` Unity Catalog discovery, respectively, both on the managed workspace | Both generated settings files retain every admin-authored default alongside the source header and every independently fetched catalog model with its label; MPS pickers keep family shortcut rows separate from catalog entries; only UC Opus/Sonnet family ids gain `[1m]` | | `test_unmanaged_claude_preserves_preexisting_family_defaults` | Seed Claude's OS-managed family defaults, then configure against one real workspace verified to have no managed config | Every pre-existing Claude family default remains unchanged in the OS-managed settings file | @@ -232,9 +235,9 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_ug_and_ucode_auth_helpers_emit_only_the_supplied_bearer` | Run both auth helper commands with the public bearer override, with and without forced refresh | Exact token-only stdout, no warnings or ANSI escapes; no workspace authentication or saved state | | `test_ug_and_ucode_web_search_helpers_preserve_mcp_stdio` | Initialize and list tools through both web-search helper commands | Exactly the MCP JSON-RPC responses; no text/ANSI contamination; existing server/tool identities preserved; no model request | -With Claude and Codex selected there are **122 live cases** (14 marked TUI cases), +With Claude and Codex selected there are **86 live cases** (14 marked TUI cases), **1 two-workspace case** (marker `workspace_switch`), -**42 managed-fixture cases** (marker `managed_fixture`, with only +**38 managed-fixture cases** (marker `managed_fixture`, with only the CodingAgentConfig input injected from a JSON file in `fixtures/managed_config/`), and **7 installation checks**. The 14 retained numbered scenarios comprise **24 explicit journeys**: 12 managed configured/fresh executions and 12 unmanaged executions. The remaining managed-fixture cases cover focused model, MCP, skills, @@ -296,7 +299,7 @@ dependency graph to reproduce a user's combination. Every relevant same-reposito PR and push to `main` runs both smoke and the full CUJ suite. Smoke covers the Databricks Hosted configure/TUI, custom OAuth CLI TUI, and headless argument journeys for both agents, in two parallel jobs. After smoke finishes, the full -suite runs all 122 live cases across two parallel agent jobs: one Claude VM and one +suite runs all 86 live cases across two parallel agent jobs: one Claude VM and one Codex VM, each running its configure, headless, and commands/lifecycle cases serially. Each agent is installed once for the full suite, and no two full jobs for the same agent overlap within a run. diff --git a/tests/integration/README.md b/tests/integration/README.md index e4e81f982..2e7192eb8 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -304,6 +304,10 @@ role-contract preservation, and isolation from legacy preference files lack dedicated regression coverage. Codex's native hook merging, project trust, and execution of pre-existing hooks are not exercised by this integration suite. +Preset parameterization is limited to routing hooks, skill toggles, explicit-model routing +bypass, and first-prompt routing. Command forwarding, app-server initialization, and catalog +fallback run without a routing selector; preset-specific behavior there is not covered. + The unit/component `../test_smart_routing_config.py` is a 135-case Cartesian oracle over all three legacy routing flags (`None`, `0`, `1`) and five selector forms (`None` plus the four supported presets, including the customer first-prompt-and-subagent mode). It independently hardcodes preset values and asserts exact @@ -395,10 +399,10 @@ startup banners and footer text cannot satisfy discovery assertions. Cases 7–1 they only configure, list models, and open/close the picker. Other live CUJs perform real model tasks. -There are **122 live cases** (including 14 marked TUI journeys) and **7 installation +There are **86 live cases** (including 14 marked TUI journeys) and **7 installation checks** with Claude and Codex; selecting OpenCode adds one live headless case. One **`workspace_switch` case** uses two real workspaces and checks skills MCP cleanup and a completed -Claude task. A further **42 `managed_fixture` cases** (six of them also `live`) run on the +Claude task. A further **38 `managed_fixture` cases** (six of them also `live`) run on the managed workspace with a checked-in JSON CodingAgentConfig from `tests/fixtures/managed_config/` injected through `UCODE_MANAGED_CONFIG_STUB`; there are no cases that read a published config. Twelve explicit configured/fresh Claude and Codex discovery and source-override journeys use the @@ -415,7 +419,7 @@ catalog discovery with overall defaults, family defaults, or both, along with ex selection and preservation of static model lists. Both cases run on the managed workspace's own bearer; no second workspace or extra secret is involved. The 14 retained numbered scenarios comprise 24 explicit journeys: 12 managed and 12 unmanaged -executions; the complete integration suite collects 103 executions. See the named coverage and gaps matrix in +executions; the complete integration suite collects 126 executions. See the named coverage and gaps matrix in [../README.md](../README.md). ```bash @@ -558,13 +562,13 @@ each test; only explicit-model scenarios choose and record a discovered Every same-repository PR and push to `main` runs **Smoke journeys**, followed by **Full journeys** even if smoke fails. Smoke runs the Hosted configure/TUI, headless argument, and custom OAuth CLI TUI journeys for each agent (six cases, -two agent jobs). Full runs all 122 live cases, including those smoke cases, in two +two agent jobs). Full runs all 86 live cases, including those smoke cases, in two disjoint agent lanes: | Agent lane | Marker | Cases | | --- | --- | --- | -| Claude | `live and claude` | 53 | -| Codex | `live and codex` | 69 | +| Claude | `live and claude` | 45 | +| Codex | `live and codex` | 41 | A non-blocking **OpenCode** job (`live and opencode`, one case) runs alongside them with `continue-on-error` and is not part of the required `cujs` gate until it is stable. @@ -637,10 +641,9 @@ Codex state comparisons exclude `.codex/tmp/arg0`, the disposable executable lin recreated by version checks, while continuing to compare persistent agent files. In addition, `test_ug_configure_managed_codex_catalog_fallback` injects the intentionally nonexistent `system.ai.gpt-99`, keeping it out of the real workspace while launching Codex through that -workspace on the valid default model `system.ai.gpt-5-6-sol`. With the selector unset and with -each supported `SMART_ROUTER_CONFIG_VERSION` preset, it opens the real Codex `/models` picker -and requires that injected custom-catalog model to be listed. The fixture itself has no managed -smart-routing setting, so the unset case remains the unconfigured baseline. +workspace on the valid default model `system.ai.gpt-5-6-sol`. It opens the real Codex `/models` +picker and requires that injected custom-catalog model to be listed. The fixture itself has +no managed smart-routing setting, and the journey does not set a routing selector. The smart-routing banner journeys inject static Claude and Codex model lists with `smart_routing` enabled in the agent config, run `ug configure`, then launch the real TUI and @@ -869,7 +872,7 @@ uv run --no-project --python 3.12 python scripts/run_integration.py \ unset DATABRICKS_BEARER ``` -This runs all 122 live cases. For the seven installation checks, run the same +This runs all 86 live cases. For the seven installation checks, run the same runner/version/index arguments with `--installation-only` and omit `-- -m live`; no bearer or workspace is needed. Results remain under `.integration-runs/`. Each invocation needs a new output directory; an existing one is rejected. diff --git a/tests/integration/test_ug_claude_commands.py b/tests/integration/test_ug_claude_commands.py index 099bc51eb..304f8f44f 100644 --- a/tests/integration/test_ug_claude_commands.py +++ b/tests/integration/test_ug_claude_commands.py @@ -5,25 +5,8 @@ pytestmark = [pytest.mark.live, pytest.mark.claude] -@pytest.mark.parametrize( - "SMART_ROUTER_CONFIG_VERSION", - [ - None, - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], - ids=[ - "unconfigured", - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], -) -def test_ug_claude_auth_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION): - """Scenario: configure claude, then ask ug for auth help under each selector. +def test_ug_claude_auth_help(live_session, workspace): + """Scenario: configure claude, then ask ug for auth help. Expected: the real claude help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -39,33 +22,14 @@ def test_ug_claude_auth_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSIO "--skip-upgrade", "--disable-databricks-ai-tools", ) - if SMART_ROUTER_CONFIG_VERSION is not None: - session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION expected = session.run("auth", "--help", binary="claude").stdout.strip() actual = session.run("claude", "--", "auth", "--help").stdout assert expected and expected in actual, actual session.assert_not_routed() -@pytest.mark.parametrize( - "SMART_ROUTER_CONFIG_VERSION", - [ - None, - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], - ids=[ - "unconfigured", - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], -) -def test_ug_claude_mcp_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION): - """Scenario: configure claude, then ask ug for mcp help under each selector. +def test_ug_claude_mcp_help(live_session, workspace): + """Scenario: configure claude, then ask ug for mcp help. Expected: the real claude help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -81,8 +45,6 @@ def test_ug_claude_mcp_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION "--skip-upgrade", "--disable-databricks-ai-tools", ) - if SMART_ROUTER_CONFIG_VERSION is not None: - session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION expected = session.run("mcp", "--help", binary="claude").stdout.strip() actual = session.run("claude", "--", "mcp", "--help").stdout assert expected and expected in actual, actual diff --git a/tests/integration/test_ug_codex_app_server.py b/tests/integration/test_ug_codex_app_server.py index 0b791b2ea..6d6b1a59e 100644 --- a/tests/integration/test_ug_codex_app_server.py +++ b/tests/integration/test_ug_codex_app_server.py @@ -6,27 +6,8 @@ @pytest.mark.parametrize("separator", [False, True], ids=["direct", "launcher-separator"]) -@pytest.mark.parametrize( - "SMART_ROUTER_CONFIG_VERSION", - [ - None, - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], - ids=[ - "unconfigured", - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], -) -def test_ug_codex_app_server_client_initializes( - live_session, workspace, separator, SMART_ROUTER_CONFIG_VERSION -): - """Scenario: configure Codex and connect a real stdio client under each selector. +def test_ug_codex_app_server_client_initializes(live_session, workspace, separator): + """Scenario: configure Codex and connect a real stdio client. Expected: initialize returns a valid JSON-RPC result, diagnostics stay off the protocol stream, and this utility command never starts smart routing. @@ -42,9 +23,6 @@ def test_ug_codex_app_server_client_initializes( "--skip-upgrade", "--disable-databricks-ai-tools", ) - if SMART_ROUTER_CONFIG_VERSION is not None: - session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION - args = ["app-server", "--listen", "stdio://"] if separator: args.insert(0, "--") diff --git a/tests/integration/test_ug_codex_commands.py b/tests/integration/test_ug_codex_commands.py index 10adff908..c7d212c00 100644 --- a/tests/integration/test_ug_codex_commands.py +++ b/tests/integration/test_ug_codex_commands.py @@ -5,25 +5,8 @@ pytestmark = [pytest.mark.live, pytest.mark.codex] -@pytest.mark.parametrize( - "SMART_ROUTER_CONFIG_VERSION", - [ - None, - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], - ids=[ - "unconfigured", - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], -) -def test_ug_codex_app_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION): - """Scenario: configure codex, then ask ug for app subcommand help under each selector. +def test_ug_codex_app_help(live_session, workspace): + """Scenario: configure codex, then ask ug for app subcommand help. Expected: the real codex help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -39,33 +22,14 @@ def test_ug_codex_app_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION) "--skip-upgrade", "--disable-databricks-ai-tools", ) - if SMART_ROUTER_CONFIG_VERSION is not None: - session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION expected = session.run("app", "--help", binary="codex").stdout.strip() actual = session.run("codex", "--", "app", "--help").stdout assert expected and expected in actual, actual session.assert_not_routed() -@pytest.mark.parametrize( - "SMART_ROUTER_CONFIG_VERSION", - [ - None, - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], - ids=[ - "unconfigured", - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], -) -def test_ug_codex_app_server_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION): - """Scenario: configure codex, then ask ug for app-server help under each selector. +def test_ug_codex_app_server_help(live_session, workspace): + """Scenario: configure codex, then ask ug for app-server help. Expected: the real codex help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -81,33 +45,14 @@ def test_ug_codex_app_server_help(live_session, workspace, SMART_ROUTER_CONFIG_V "--skip-upgrade", "--disable-databricks-ai-tools", ) - if SMART_ROUTER_CONFIG_VERSION is not None: - session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION expected = session.run("app-server", "--help", binary="codex").stdout.strip() actual = session.run("codex", "--", "app-server", "--help").stdout assert expected and expected in actual, actual session.assert_not_routed() -@pytest.mark.parametrize( - "SMART_ROUTER_CONFIG_VERSION", - [ - None, - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], - ids=[ - "unconfigured", - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], -) -def test_ug_codex_exec_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION): - """Scenario: configure codex, then ask ug for exec help under each selector. +def test_ug_codex_exec_help(live_session, workspace): + """Scenario: configure codex, then ask ug for exec help. Expected: the real codex help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -123,33 +68,14 @@ def test_ug_codex_exec_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION "--skip-upgrade", "--disable-databricks-ai-tools", ) - if SMART_ROUTER_CONFIG_VERSION is not None: - session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION expected = session.run("exec", "--help", binary="codex").stdout.strip() actual = session.run("codex", "--", "exec", "--help").stdout assert expected and expected in actual, actual session.assert_not_routed() -@pytest.mark.parametrize( - "SMART_ROUTER_CONFIG_VERSION", - [ - None, - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], - ids=[ - "unconfigured", - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], -) -def test_ug_codex_mcp_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION): - """Scenario: configure codex, then ask ug for mcp help under each selector. +def test_ug_codex_mcp_help(live_session, workspace): + """Scenario: configure codex, then ask ug for mcp help. Expected: the real codex help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -165,35 +91,14 @@ def test_ug_codex_mcp_help(live_session, workspace, SMART_ROUTER_CONFIG_VERSION) "--skip-upgrade", "--disable-databricks-ai-tools", ) - if SMART_ROUTER_CONFIG_VERSION is not None: - session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION expected = session.run("mcp", "--help", binary="codex").stdout.strip() actual = session.run("codex", "--", "mcp", "--help").stdout assert expected and expected in actual, actual session.assert_not_routed() -@pytest.mark.parametrize( - "SMART_ROUTER_CONFIG_VERSION", - [ - None, - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], - ids=[ - "unconfigured", - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], -) -def test_ug_codex_app_reports_unknown_argument( - live_session, workspace, SMART_ROUTER_CONFIG_VERSION -): - """Scenario: pass an unknown option directly to ug codex app under each selector. +def test_ug_codex_app_reports_unknown_argument(live_session, workspace): + """Scenario: pass an unknown option directly to ug codex app. Expected: the actual Codex parser's error and exit status are preserved, without opening a desktop application or starting routing. @@ -209,8 +114,6 @@ def test_ug_codex_app_reports_unknown_argument( "--skip-upgrade", "--disable-databricks-ai-tools", ) - if SMART_ROUTER_CONFIG_VERSION is not None: - session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION args = ["app", "--ug-integration-unknown-option"] expected = session.run(*args, binary="codex", ok=False) actual = session.run("codex", *args, ok=False) diff --git a/tests/integration/test_ug_configure_managed_models.py b/tests/integration/test_ug_configure_managed_models.py index b4c6e2fe5..76b5be203 100644 --- a/tests/integration/test_ug_configure_managed_models.py +++ b/tests/integration/test_ug_configure_managed_models.py @@ -171,31 +171,11 @@ def test_managed_fixture_claude_model_picker_reflects_the_config(live_session, w @pytest.mark.managed_fixture @pytest.mark.codex -@pytest.mark.parametrize( - "SMART_ROUTER_CONFIG_VERSION", - [ - None, - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], - ids=[ - "unconfigured", - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], -) -def test_ug_configure_managed_codex_catalog_fallback( - live_session, workspace, SMART_ROUTER_CONFIG_VERSION -): - """Scenario: configure Codex from an injected model list under each selector. +def test_ug_configure_managed_codex_catalog_fallback(live_session, workspace): + """Scenario: configure Codex from an injected model list. Expected: ug creates conservative fallback metadata for the unknown model, warns how to get - richer metadata, and the real Codex TUI lists that model in its /model picker for the - unconfigured case and each supported routing selector. + richer metadata, and the real Codex TUI lists that model in its /model picker. """ session = live_session use_managed_config_fixture(session, "codex_catalog_fallback") @@ -217,13 +197,11 @@ def test_ug_configure_managed_codex_catalog_fallback( assert fallback.get("context_window") == 32768, fallback assert fallback.get("default_reasoning_level") == "none", fallback - if SMART_ROUTER_CONFIG_VERSION is not None: - session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION with AgentTerminal( session, "codex", [str(session.binary), "codex"], - f"managed-fallback-routing-{SMART_ROUTER_CONFIG_VERSION or 'unconfigured'}", + "managed-fallback", ) as tui: tui.boot() tui.submit("/model") From 31819bf16d1be26ffece0b508ca0b0ddc7b049cc Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Thu, 8 Oct 2026 23:13:39 +0000 Subject: [PATCH 12/30] Preserve legacy integration cases alongside smart router presets --- tests/README.md | 38 ++++--- tests/integration/README.md | 35 +++--- tests/integration/test_ug_claude_commands.py | 12 +- tests/integration/test_ug_claude_headless.py | 23 +--- tests/integration/test_ug_codex_app_server.py | 7 +- tests/integration/test_ug_codex_commands.py | 26 +++-- tests/integration/test_ug_codex_headless.py | 23 +--- .../test_ug_configure_managed_models.py | 14 +-- .../test_ug_smart_routing_hooks.py | 104 +++++++++++++----- 9 files changed, 157 insertions(+), 125 deletions(-) diff --git a/tests/README.md b/tests/README.md index 4e66e3c9a..1adfdc327 100644 --- a/tests/README.md +++ b/tests/README.md @@ -145,17 +145,19 @@ CLI startup ordering, snapshots/restoration, routing getters, native subcommands session files, or managed launches are not established by this grid. These are component checks; they do not establish live agent, hook, or gateway behavior. -The toggle integration journeys intentionally run with only the three subagent-only -`SMART_ROUTER_CONFIG_VERSION` presets. They require only `smart-router` for `subagent_only_v0` -and `subagent_only_v1`, and both bundled skills for `subagent_orch_v0`; they verify the saved +The toggle integration journeys retain the legacy routing-only and orchestration cases and +add the three subagent-only `SMART_ROUTER_CONFIG_VERSION` presets. They require only +`smart-router` for routing-only cases, and both bundled skills when orchestration is enabled; +they verify the saved session controls and native tool-result confirmation after each toggle, and explicitly request their children, including while routing is off. The managed-fixture banner journeys separately cover the selector unset (managed default) and `first_prompt_and_subagent_no_orch_v0` for real first-prompt tasks. -Preset parameterization is limited to routing hooks, skill toggles, explicit-model routing -bypass, and first-prompt routing. Command forwarding, app-server initialization, and catalog -fallback run without a routing selector; preset-specific behavior there is not covered. +Preset parameterization augments only routing hooks, skill toggles, and first-prompt routing; +their original legacy-env or managed-default cases remain. Explicit-model selection, command +forwarding, app-server initialization, and catalog fallback retain their original legacy flags +and on/off coverage; preset-specific behavior there is not covered. `test_integration_evidence.py` checks native tool-result extraction for both agents, including collapsed-output records, and excludes user echoes and assistant claims. @@ -205,16 +207,16 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_ug_codex_headless_prompt_argument`, `test_ug_codex_headless_prompt_stdin`, `test_ug_codex_headless_prompt_after_separator` | Run Codex from a script using each prompt form | Completed turn and final answer contain the file value; exit zero; no routing | | `test_ug_opencode_headless_prompt_argument` | Run OpenCode from a script (`run --format json --auto`) with an argument prompt | Completed Read tool call; final text answer contains the file value; exit zero (non-blocking CI lane) | | `test_ug_claude_exports_trace_to_configured_table`, `test_ug_codex_exports_trace_to_configured_table` | Configure tracing, complete a headless task carrying a unique trace marker, then wait for ingestion | The configured trace table contains an agent span with the same trace-safe marker and requested model | -| `test_ug_claude_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` before and after ug's separator under each supported routing preset, without workspace policy | Real file task completes; JSON `modelUsage` reports the requested model with output tokens; no routing wrapper | -| `test_ug_codex_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` / `-m VALUE` under each supported routing preset | Real file task completes; no routing wrapper | +| `test_ug_claude_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` before and after ug's separator with smart routing enabled, without workspace policy | Real file task completes; JSON `modelUsage` reports the requested model with output tokens; no routing wrapper | +| `test_ug_codex_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` / `-m VALUE` with smart routing enabled | Real file task completes; no routing wrapper | | `test_ug_claude_preserves_caller_settings_and_hook` | Pass a settings path containing spaces | Real SessionStart hook executes; caller file unchanged; file task completes | | `test_ug_claude_reports_unsupported_short_model_option` | Pass Claude's unsupported `-m` | Actual agent error and exit status preserved | -| `test_ug_claude_auth_help`, `test_ug_claude_mcp_help` | Request subcommand help | Real agent help; no routing wrapper | -| `test_ug_codex_app_help`, `test_ug_codex_app_server_help`, `test_ug_codex_exec_help`, `test_ug_codex_mcp_help` | Request subcommand help | Real agent help; no routing wrapper | -| `test_ug_codex_app_reports_unknown_argument` | Pass an invalid option directly to `ug codex app` | Real Codex parser error and status preserved | -| `test_ug_codex_app_server_client_initializes` | Connect a stdio client, direct/`--` separator | Actual JSON-RPC initialize response; no non-JSON stdout; no routing | -| `test_smart_routing_claude_route_subagent_hook`, `test_smart_routing_codex_route_subagent_hook` | Pipe a real PreToolUse spawn payload to the installed route-subagent hook under each supported routing preset | Allow decision against the live router; requested model replaced by a routed agent definition (Claude) or bundled catalog slug (Codex) from the offered models; one audited decision matching the session and task | -| `test_smart_router_skill_toggles_claude_subagent_routing`, `test_smart_router_skill_toggles_codex_subagent_routing` | Configure, launch a real subagent-only TUI under the three subagent-only presets, then spawn tagged children while invoking the installed Smart Router skill to switch routing on -> off -> on in the same session | `subagent_only_v0` and `subagent_only_v1` install `smart-router`; `subagent_orch_v0` also installs `smart-router-orchestrator`; all three native children complete; only routing-enabled phases show the subagent banner and produce a live routing decision correlated with the child; no first-prompt routing wrapper; the customer full-mode preset is intentionally outside this journey; normal exit | +| `test_ug_claude_auth_help`, `test_ug_claude_mcp_help` | Request subcommand help with legacy smart routing off/on | Real agent help; no routing wrapper | +| `test_ug_codex_app_help`, `test_ug_codex_app_server_help`, `test_ug_codex_exec_help`, `test_ug_codex_mcp_help` | Request subcommand help with legacy smart routing off/on | Real agent help; no routing wrapper | +| `test_ug_codex_app_reports_unknown_argument` | Pass an invalid option directly to `ug codex app` with legacy smart routing off/on | Real Codex parser error and status preserved | +| `test_ug_codex_app_server_client_initializes` | Connect a stdio client, direct/`--` separator, with legacy smart routing off/on | Actual JSON-RPC initialize response; no non-JSON stdout; no routing | +| `test_smart_routing_claude_route_subagent_hook`, `test_smart_routing_codex_route_subagent_hook` | Pipe a real PreToolUse spawn payload to the installed route-subagent hook with legacy subagent routing or each supported routing preset | Allow decision against the live router; requested model replaced by a routed agent definition (Claude) or bundled catalog slug (Codex) from the offered models; one audited decision matching the session and task | +| `test_smart_router_skill_toggles_claude_subagent_routing`, `test_smart_router_skill_toggles_codex_subagent_routing` | Configure, launch a real subagent-only TUI with legacy routing-only/orchestration flags or the three subagent-only presets, then spawn tagged children while invoking the installed Smart Router skill to switch routing on -> off -> on in the same session | Routing-only cases install `smart-router`; orchestration cases also install `smart-router-orchestrator`; all three native children complete; only routing-enabled phases show the subagent banner and produce a live routing decision correlated with the child; no first-prompt routing wrapper; the customer full-mode preset is intentionally outside this journey; normal exit | | `test_ug_configure_claude_repeat_and_revert`, `test_ug_configure_codex_repeat_and_revert` | Configure twice over user settings; complete a task; revert twice | Settings preserved; no bearer in ug state; generated config removed; status unconfigured | | `test_ug_configure_claude_cleans_stale_skills_mcp_on_workspace_switch` | Configure the first workspace, register its skills MCP, switch to a second real workspace, and use Claude | Old registration removed from Claude and the new workspace state; old workspace bucket preserved; repeat configure stays clean; real file task completes on the second workspace | | `test_ug_configure_claude_rejects_invalid_credentials`, `test_ug_configure_codex_rejects_invalid_credentials` | Configure with a rejected bearer against the real workspace | Authentication failure; no successful saved setup | @@ -224,7 +226,7 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_case_03_*`, `test_case_05_*` | Pass a provider or model-location override to managed Claude after configure and from fresh state | ug rejects the override before Claude starts and preserves agent-owned state | | `test_case_02_*` | Launch managed Codex after configure and from fresh state | The scoped and stable catalogs, ug-launched app server, and fresh bare app server match the independently fetched admin MPS model IDs. The configured case uses real `ug revert` to remove ug's shared pointer and stable file while preserving a user setting | | `test_case_04_*`, `test_case_06_*` | Pass a provider or model-location override to managed Codex after configure and from fresh state | ug rejects the override before Codex starts and preserves agent-owned state | -| `test_ug_configure_managed_codex_catalog_fallback` | Configure from an injected managed response containing a GPT model absent from Codex's bundled catalog, then inspect the picker | Actionable metadata warning; conservative catalog entry for the unknown model; real Codex picker lists the custom catalog model | +| `test_ug_configure_managed_codex_catalog_fallback` | Configure from an injected managed response containing a GPT model absent from Codex's bundled catalog, then inspect the picker with legacy smart routing off/on | Actionable metadata warning; conservative catalog entry for the unknown model; real Codex picker lists the custom catalog model | | `test_managed_fixture_codex_http_headers_in_managed_file` | Interactive PTY configure with injected managed `http_headers` for Codex | The specified header (`x-databricks-workspace`) lands in `model_providers.Databricks.http_headers` in `/etc/codex/managed_config.toml` with the exact admin value | | `test_managed_claude_mps_defaults_accompany_discovery`, `test_managed_claude_parent_schema_defaults_accompany_discovery` | Configure from a stubbed config and launch Claude with MPS discovery (`main.default.ci_e2e_anthropic_mps`) and with `system.ai` Unity Catalog discovery, respectively, both on the managed workspace | Both generated settings files retain every admin-authored default alongside the source header and every independently fetched catalog model with its label; MPS pickers keep family shortcut rows separate from catalog entries; only UC Opus/Sonnet family ids gain `[1m]` | | `test_unmanaged_claude_preserves_preexisting_family_defaults` | Seed Claude's OS-managed family defaults, then configure against one real workspace verified to have no managed config | Every pre-existing Claude family default remains unchanged in the OS-managed settings file | @@ -235,9 +237,9 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_ug_and_ucode_auth_helpers_emit_only_the_supplied_bearer` | Run both auth helper commands with the public bearer override, with and without forced refresh | Exact token-only stdout, no warnings or ANSI escapes; no workspace authentication or saved state | | `test_ug_and_ucode_web_search_helpers_preserve_mcp_stdio` | Initialize and list tools through both web-search helper commands | Exactly the MCP JSON-RPC responses; no text/ANSI contamination; existing server/tool identities preserved; no model request | -With Claude and Codex selected there are **86 live cases** (14 marked TUI cases), +With Claude and Codex selected there are **80 live cases** (14 marked TUI cases), **1 two-workspace case** (marker `workspace_switch`), -**38 managed-fixture cases** (marker `managed_fixture`, with only +**43 managed-fixture cases** (marker `managed_fixture`, with only the CodingAgentConfig input injected from a JSON file in `fixtures/managed_config/`), and **7 installation checks**. The 14 retained numbered scenarios comprise **24 explicit journeys**: 12 managed configured/fresh executions and 12 unmanaged executions. The remaining managed-fixture cases cover focused model, MCP, skills, @@ -299,7 +301,7 @@ dependency graph to reproduce a user's combination. Every relevant same-reposito PR and push to `main` runs both smoke and the full CUJ suite. Smoke covers the Databricks Hosted configure/TUI, custom OAuth CLI TUI, and headless argument journeys for both agents, in two parallel jobs. After smoke finishes, the full -suite runs all 86 live cases across two parallel agent jobs: one Claude VM and one +suite runs all 80 live cases across two parallel agent jobs: one Claude VM and one Codex VM, each running its configure, headless, and commands/lifecycle cases serially. Each agent is installed once for the full suite, and no two full jobs for the same agent overlap within a run. diff --git a/tests/integration/README.md b/tests/integration/README.md index 2e7192eb8..7716dcc08 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -289,10 +289,10 @@ PATH conflicts for the Smart Router skill have subprocess/component coverage in `ug` first in PATH. The live journeys above do not inject a second installation or establish PowerShell command execution. -The toggle journeys intentionally run once under each of the three subagent-only -`SMART_ROUTER_CONFIG_VERSION` presets. They require only `smart-router` for `subagent_only_v0` -and `subagent_only_v1`, and both `smart-router-orchestrator` and `smart-router` for -`subagent_orch_v0`. They verify the saved session controls, a new CLI confirmation in the native +The toggle journeys retain the legacy routing-only and orchestration cases and add each of +the three subagent-only `SMART_ROUTER_CONFIG_VERSION` presets. They require only `smart-router` +for routing-only cases, and both `smart-router-orchestrator` and `smart-router` when orchestration +is enabled. They verify the saved session controls, a new CLI confirmation in the native tool-result records, and a new assistant answer after each skill invocation. Collapsed terminal output is allowed; the answer need not repeat the CLI's exact wording. @@ -304,9 +304,10 @@ role-contract preservation, and isolation from legacy preference files lack dedicated regression coverage. Codex's native hook merging, project trust, and execution of pre-existing hooks are not exercised by this integration suite. -Preset parameterization is limited to routing hooks, skill toggles, explicit-model routing -bypass, and first-prompt routing. Command forwarding, app-server initialization, and catalog -fallback run without a routing selector; preset-specific behavior there is not covered. +Preset parameterization augments only routing hooks, skill toggles, and first-prompt routing; +their original legacy-env or managed-default cases remain. Explicit-model selection, command +forwarding, app-server initialization, and catalog fallback retain their original legacy flags +and on/off coverage; preset-specific behavior there is not covered. The unit/component `../test_smart_routing_config.py` is a 135-case Cartesian oracle over all three legacy routing flags (`None`, `0`, `1`) and five selector forms (`None` plus the four @@ -399,10 +400,10 @@ startup banners and footer text cannot satisfy discovery assertions. Cases 7–1 they only configure, list models, and open/close the picker. Other live CUJs perform real model tasks. -There are **86 live cases** (including 14 marked TUI journeys) and **7 installation +There are **80 live cases** (including 14 marked TUI journeys) and **7 installation checks** with Claude and Codex; selecting OpenCode adds one live headless case. One **`workspace_switch` case** uses two real workspaces and checks skills MCP cleanup and a completed -Claude task. A further **38 `managed_fixture` cases** (six of them also `live`) run on the +Claude task. A further **43 `managed_fixture` cases** (ten of them also `live`) run on the managed workspace with a checked-in JSON CodingAgentConfig from `tests/fixtures/managed_config/` injected through `UCODE_MANAGED_CONFIG_STUB`; there are no cases that read a published config. Twelve explicit configured/fresh Claude and Codex discovery and source-override journeys use the @@ -419,7 +420,7 @@ catalog discovery with overall defaults, family defaults, or both, along with ex selection and preservation of static model lists. Both cases run on the managed workspace's own bearer; no second workspace or extra secret is involved. The 14 retained numbered scenarios comprise 24 explicit journeys: 12 managed and 12 unmanaged -executions; the complete integration suite collects 126 executions. See the named coverage and gaps matrix in +executions; the complete integration suite collects 121 executions. See the named coverage and gaps matrix in [../README.md](../README.md). ```bash @@ -562,13 +563,13 @@ each test; only explicit-model scenarios choose and record a discovered Every same-repository PR and push to `main` runs **Smoke journeys**, followed by **Full journeys** even if smoke fails. Smoke runs the Hosted configure/TUI, headless argument, and custom OAuth CLI TUI journeys for each agent (six cases, -two agent jobs). Full runs all 86 live cases, including those smoke cases, in two +two agent jobs). Full runs all 80 live cases, including those smoke cases, in two disjoint agent lanes: | Agent lane | Marker | Cases | | --- | --- | --- | -| Claude | `live and claude` | 45 | -| Codex | `live and codex` | 41 | +| Claude | `live and claude` | 38 | +| Codex | `live and codex` | 42 | A non-blocking **OpenCode** job (`live and opencode`, one case) runs alongside them with `continue-on-error` and is not part of the required `cujs` gate until it is stable. @@ -641,9 +642,9 @@ Codex state comparisons exclude `.codex/tmp/arg0`, the disposable executable lin recreated by version checks, while continuing to compare persistent agent files. In addition, `test_ug_configure_managed_codex_catalog_fallback` injects the intentionally nonexistent `system.ai.gpt-99`, keeping it out of the real workspace while launching Codex through that -workspace on the valid default model `system.ai.gpt-5-6-sol`. It opens the real Codex `/models` -picker and requires that injected custom-catalog model to be listed. The fixture itself has -no managed smart-routing setting, and the journey does not set a routing selector. +workspace on the valid default model `system.ai.gpt-5-6-sol`. With legacy smart routing off and +on, it opens the real Codex `/models` picker and requires that injected custom-catalog model +to be listed. The fixture itself has no managed smart-routing setting. The smart-routing banner journeys inject static Claude and Codex model lists with `smart_routing` enabled in the agent config, run `ug configure`, then launch the real TUI and @@ -872,7 +873,7 @@ uv run --no-project --python 3.12 python scripts/run_integration.py \ unset DATABRICKS_BEARER ``` -This runs all 86 live cases. For the seven installation checks, run the same +This runs all 80 live cases. For the seven installation checks, run the same runner/version/index arguments with `--installation-only` and omit `-- -m live`; no bearer or workspace is needed. Results remain under `.integration-runs/`. Each invocation needs a new output directory; an existing one is rejected. diff --git a/tests/integration/test_ug_claude_commands.py b/tests/integration/test_ug_claude_commands.py index 304f8f44f..701c624ab 100644 --- a/tests/integration/test_ug_claude_commands.py +++ b/tests/integration/test_ug_claude_commands.py @@ -5,8 +5,9 @@ pytestmark = [pytest.mark.live, pytest.mark.claude] -def test_ug_claude_auth_help(live_session, workspace): - """Scenario: configure claude, then ask ug for auth help. +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_claude_auth_help(live_session, workspace, routing): + """Scenario: configure claude, then ask ug for auth subcommand help. Expected: the real claude help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -22,14 +23,16 @@ def test_ug_claude_auth_help(live_session, workspace): "--skip-upgrade", "--disable-databricks-ai-tools", ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing expected = session.run("auth", "--help", binary="claude").stdout.strip() actual = session.run("claude", "--", "auth", "--help").stdout assert expected and expected in actual, actual session.assert_not_routed() -def test_ug_claude_mcp_help(live_session, workspace): - """Scenario: configure claude, then ask ug for mcp help. +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_claude_mcp_help(live_session, workspace, routing): + """Scenario: configure claude, then ask ug for mcp subcommand help. Expected: the real claude help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -45,6 +48,7 @@ def test_ug_claude_mcp_help(live_session, workspace): "--skip-upgrade", "--disable-databricks-ai-tools", ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing expected = session.run("mcp", "--help", binary="claude").stdout.strip() actual = session.run("claude", "--", "mcp", "--help").stdout assert expected and expected in actual, actual diff --git a/tests/integration/test_ug_claude_headless.py b/tests/integration/test_ug_claude_headless.py index bd83d3b00..4a41204de 100644 --- a/tests/integration/test_ug_claude_headless.py +++ b/tests/integration/test_ug_claude_headless.py @@ -116,27 +116,12 @@ def test_ug_claude_headless_prompt_after_separator(live_session, workspace): @pytest.mark.parametrize("model_form", ["separate", "equals"]) @pytest.mark.parametrize("model_owner", ["ug", "claude"]) -@pytest.mark.parametrize( - "SMART_ROUTER_CONFIG_VERSION", - [ - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], - ids=[ - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], -) def test_ug_claude_headless_explicit_model_bypasses_routing( - live_session, workspace, model_form, model_owner, SMART_ROUTER_CONFIG_VERSION + live_session, workspace, model_form, model_owner ): - """Scenario: choose a model before/after ug's separator under each supported selector. + """Scenario: choose a model before/after ug's separator with no workspace policy. - Expected: under each supported routing preset, the real file task completes on the + Expected: with smart routing enabled, the real file task completes on the requested model, confirmed by JSON modelUsage, without a routing wrapper. """ session = live_session @@ -153,7 +138,7 @@ def test_ug_claude_headless_explicit_model_bypasses_routing( ) model = session.model_for_explicit_case("claude") model_args = ["--model", model] if model_form == "separate" else [f"--model={model}"] - session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION + session.env["ENABLE_SMART_ROUTING_V2"] = "1" result = session.run( "claude", *(model_args if model_owner == "ug" else []), diff --git a/tests/integration/test_ug_codex_app_server.py b/tests/integration/test_ug_codex_app_server.py index 6d6b1a59e..9845dd190 100644 --- a/tests/integration/test_ug_codex_app_server.py +++ b/tests/integration/test_ug_codex_app_server.py @@ -6,8 +6,9 @@ @pytest.mark.parametrize("separator", [False, True], ids=["direct", "launcher-separator"]) -def test_ug_codex_app_server_client_initializes(live_session, workspace, separator): - """Scenario: configure Codex and connect a real stdio client. +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_app_server_client_initializes(live_session, workspace, separator, routing): + """Scenario: configure Codex and connect a real stdio client to ug codex app-server. Expected: initialize returns a valid JSON-RPC result, diagnostics stay off the protocol stream, and this utility command never starts smart routing. @@ -23,6 +24,8 @@ def test_ug_codex_app_server_client_initializes(live_session, workspace, separat "--skip-upgrade", "--disable-databricks-ai-tools", ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + args = ["app-server", "--listen", "stdio://"] if separator: args.insert(0, "--") diff --git a/tests/integration/test_ug_codex_commands.py b/tests/integration/test_ug_codex_commands.py index c7d212c00..fe382fba5 100644 --- a/tests/integration/test_ug_codex_commands.py +++ b/tests/integration/test_ug_codex_commands.py @@ -5,7 +5,8 @@ pytestmark = [pytest.mark.live, pytest.mark.codex] -def test_ug_codex_app_help(live_session, workspace): +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_app_help(live_session, workspace, routing): """Scenario: configure codex, then ask ug for app subcommand help. Expected: the real codex help is returned, with no routing wrapper or @@ -22,14 +23,16 @@ def test_ug_codex_app_help(live_session, workspace): "--skip-upgrade", "--disable-databricks-ai-tools", ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing expected = session.run("app", "--help", binary="codex").stdout.strip() actual = session.run("codex", "--", "app", "--help").stdout assert expected and expected in actual, actual session.assert_not_routed() -def test_ug_codex_app_server_help(live_session, workspace): - """Scenario: configure codex, then ask ug for app-server help. +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_app_server_help(live_session, workspace, routing): + """Scenario: configure codex, then ask ug for app-server subcommand help. Expected: the real codex help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -45,14 +48,16 @@ def test_ug_codex_app_server_help(live_session, workspace): "--skip-upgrade", "--disable-databricks-ai-tools", ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing expected = session.run("app-server", "--help", binary="codex").stdout.strip() actual = session.run("codex", "--", "app-server", "--help").stdout assert expected and expected in actual, actual session.assert_not_routed() -def test_ug_codex_exec_help(live_session, workspace): - """Scenario: configure codex, then ask ug for exec help. +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_exec_help(live_session, workspace, routing): + """Scenario: configure codex, then ask ug for exec subcommand help. Expected: the real codex help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -68,14 +73,16 @@ def test_ug_codex_exec_help(live_session, workspace): "--skip-upgrade", "--disable-databricks-ai-tools", ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing expected = session.run("exec", "--help", binary="codex").stdout.strip() actual = session.run("codex", "--", "exec", "--help").stdout assert expected and expected in actual, actual session.assert_not_routed() -def test_ug_codex_mcp_help(live_session, workspace): - """Scenario: configure codex, then ask ug for mcp help. +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_mcp_help(live_session, workspace, routing): + """Scenario: configure codex, then ask ug for mcp subcommand help. Expected: the real codex help is returned, with no routing wrapper or model request. This verifies command dispatch, not an interactive session. @@ -91,13 +98,15 @@ def test_ug_codex_mcp_help(live_session, workspace): "--skip-upgrade", "--disable-databricks-ai-tools", ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing expected = session.run("mcp", "--help", binary="codex").stdout.strip() actual = session.run("codex", "--", "mcp", "--help").stdout assert expected and expected in actual, actual session.assert_not_routed() -def test_ug_codex_app_reports_unknown_argument(live_session, workspace): +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_app_reports_unknown_argument(live_session, workspace, routing): """Scenario: pass an unknown option directly to ug codex app. Expected: the actual Codex parser's error and exit status are preserved, @@ -114,6 +123,7 @@ def test_ug_codex_app_reports_unknown_argument(live_session, workspace): "--skip-upgrade", "--disable-databricks-ai-tools", ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing args = ["app", "--ug-integration-unknown-option"] expected = session.run(*args, binary="codex", ok=False) actual = session.run("codex", *args, ok=False) diff --git a/tests/integration/test_ug_codex_headless.py b/tests/integration/test_ug_codex_headless.py index 2a9378337..b58f5dcd8 100644 --- a/tests/integration/test_ug_codex_headless.py +++ b/tests/integration/test_ug_codex_headless.py @@ -119,25 +119,8 @@ def test_ug_codex_headless_prompt_after_separator(live_session, workspace): @pytest.mark.parametrize("model_form", ["separate", "equals", "short"]) -@pytest.mark.parametrize( - "SMART_ROUTER_CONFIG_VERSION", - [ - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], - ids=[ - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", - ], -) -def test_ug_codex_headless_explicit_model_bypasses_routing( - live_session, workspace, model_form, SMART_ROUTER_CONFIG_VERSION -): - """Scenario: choose an explicit model under each supported routing selector. +def test_ug_codex_headless_explicit_model_bypasses_routing(live_session, workspace, model_form): + """Scenario: choose an explicit model while global smart routing is enabled. Expected: the model option is accepted, the real file task completes, and no routing wrapper overrides the caller's choice. @@ -158,7 +141,7 @@ def test_ug_codex_headless_explicit_model_bypasses_routing( model_args = ["--model", model] if model_form == "separate" else [f"--model={model}"] if model_form == "short": model_args = ["-m", model] - session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION + session.env["ENABLE_SMART_ROUTING_V2"] = "1" result = session.run( "codex", "--", diff --git a/tests/integration/test_ug_configure_managed_models.py b/tests/integration/test_ug_configure_managed_models.py index 76b5be203..8433617cd 100644 --- a/tests/integration/test_ug_configure_managed_models.py +++ b/tests/integration/test_ug_configure_managed_models.py @@ -171,11 +171,13 @@ def test_managed_fixture_claude_model_picker_reflects_the_config(live_session, w @pytest.mark.managed_fixture @pytest.mark.codex -def test_ug_configure_managed_codex_catalog_fallback(live_session, workspace): - """Scenario: configure Codex from an injected model list. +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_configure_managed_codex_catalog_fallback(live_session, workspace, routing): + """Scenario: configure Codex from an injected model list containing an unknown GPT model. Expected: ug creates conservative fallback metadata for the unknown model, warns how to get - richer metadata, and the real Codex TUI lists that model in its /model picker. + richer metadata, and the real Codex TUI lists that model in its /model picker both with and + without smart routing. """ session = live_session use_managed_config_fixture(session, "codex_catalog_fallback") @@ -197,11 +199,9 @@ def test_ug_configure_managed_codex_catalog_fallback(live_session, workspace): assert fallback.get("context_window") == 32768, fallback assert fallback.get("default_reasoning_level") == "none", fallback + session.env["ENABLE_SMART_ROUTING_V2"] = routing with AgentTerminal( - session, - "codex", - [str(session.binary), "codex"], - "managed-fallback", + session, "codex", [str(session.binary), "codex"], f"managed-fallback-routing-{routing}" ) as tui: tui.boot() tui.submit("/model") diff --git a/tests/integration/test_ug_smart_routing_hooks.py b/tests/integration/test_ug_smart_routing_hooks.py index 0515efae6..3f7505774 100644 --- a/tests/integration/test_ug_smart_routing_hooks.py +++ b/tests/integration/test_ug_smart_routing_hooks.py @@ -112,7 +112,7 @@ def _run_calculation(tui, session, agent: str, expression: str, expected: str, * def _toggle_with_skill( - tui, session, agent: str, enabled: bool, SMART_ROUTER_CONFIG_VERSION: str + tui, session, agent: str, enabled: bool, orchestration_enabled: bool ) -> None: skill_root = session.home / SKILL_ROOTS[agent] ignored_skills = {".system"} if agent == "codex" else set() @@ -122,9 +122,7 @@ def _toggle_with_skill( if path.is_dir() and path.name not in ignored_skills ) expected_skills = ( - ["smart-router", "smart-router-orchestrator"] - if (SMART_ROUTER_CONFIG_VERSION == "subagent_orch_v0") - else ["smart-router"] + ["smart-router", "smart-router-orchestrator"] if orchestration_enabled else ["smart-router"] ) assert installed_skills == expected_skills, installed_skills @@ -177,12 +175,14 @@ def toggled(_screen): @pytest.mark.parametrize( "SMART_ROUTER_CONFIG_VERSION", [ + None, "first_prompt_and_subagent_no_orch_v0", "subagent_only_v0", "subagent_only_v1", "subagent_orch_v0", ], ids=[ + "legacy-env", "first_prompt_and_subagent_no_orch_v0", "subagent_only_v0", "subagent_only_v1", @@ -192,8 +192,8 @@ def toggled(_screen): def test_smart_routing_claude_route_subagent_hook( live_session, workspace, SMART_ROUTER_CONFIG_VERSION ): - """Scenario: each supported routing selector makes Claude Code fire PreToolUse for an - Agent spawn, piping the payload to ``ug claude-router-hook route-subagent``. + """Scenario: legacy subagent routing or each supported selector makes Claude Code fire + PreToolUse for an Agent spawn, piping the payload to ``ug claude-router-hook route-subagent``. Expected: the hook allows the call against the real workspace router, drops the requested model in favor of a ``ucode-route-`` agent definition while preserving the @@ -201,7 +201,10 @@ def test_smart_routing_claude_route_subagent_hook( hook contract is asserted; no agent decides to spawn. """ session = live_session - session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION + if SMART_ROUTER_CONFIG_VERSION is None: + session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" + else: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION payload = { "session_id": "claude-route-subagent-hook", "tool_name": "Agent", @@ -248,12 +251,14 @@ def test_smart_routing_claude_route_subagent_hook( @pytest.mark.parametrize( "SMART_ROUTER_CONFIG_VERSION", [ + None, "first_prompt_and_subagent_no_orch_v0", "subagent_only_v0", "subagent_only_v1", "subagent_orch_v0", ], ids=[ + "legacy-env", "first_prompt_and_subagent_no_orch_v0", "subagent_only_v0", "subagent_only_v1", @@ -263,8 +268,8 @@ def test_smart_routing_claude_route_subagent_hook( def test_smart_routing_codex_route_subagent_hook( live_session, workspace, SMART_ROUTER_CONFIG_VERSION ): - """Scenario: each supported routing selector makes Codex fire PreToolUse for a - spawn_agent call, piping the payload to ``ug codex-router-hook route-subagent``. + """Scenario: legacy subagent routing or each supported selector makes Codex fire PreToolUse + for a spawn_agent call, piping the payload to ``ug codex-router-hook route-subagent``. Expected: the hook allows the call against the real workspace router, rewrites the requested model to the bundled catalog slug of an offered model while preserving the @@ -272,7 +277,10 @@ def test_smart_routing_codex_route_subagent_hook( hook contract is asserted; no agent decides to spawn. """ session = live_session - session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION + if SMART_ROUTER_CONFIG_VERSION is None: + session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" + else: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION payload = { "session_id": "codex-route-subagent-hook", "tool_name": "spawn_agent", @@ -316,19 +324,31 @@ def test_smart_routing_codex_route_subagent_hook( @pytest.mark.claude @pytest.mark.managed_fixture @pytest.mark.parametrize( - "SMART_ROUTER_CONFIG_VERSION", - ["subagent_only_v0", "subagent_only_v1", "subagent_orch_v0"], - ids=["subagent_only_v0", "subagent_only_v1", "subagent_orch_v0"], + "orchestration_enabled, SMART_ROUTER_CONFIG_VERSION", + [ + (False, None), + (True, None), + (False, "subagent_only_v0"), + (False, "subagent_only_v1"), + (True, "subagent_orch_v0"), + ], + ids=[ + "routing-only", + "orchestration", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], ) def test_smart_router_skill_toggles_claude_subagent_routing( - live_session, workspace, tmp_path, SMART_ROUTER_CONFIG_VERSION + live_session, workspace, tmp_path, orchestration_enabled, SMART_ROUTER_CONFIG_VERSION ): - """Scenario: launch Claude with each supported subagent selector, spawn a child, invoke + """Scenario: launch Claude with the legacy flags or a subagent selector, spawn a child, invoke the installed Smart Router skill to turn routing off, spawn another child, turn routing back on through the skill, and spawn a third child in the same real TUI session. - Expected: subagent_only_v0 and subagent_only_v1 install Smart Router, while - subagent_orch_v0 also installs Smart Router Orchestrator. Each invocation records the CLI + Expected: routing-only cases install Smart Router; enabling orchestration also installs + Smart Router Orchestrator. Each invocation records the CLI confirmation in the native transcript and changes the saved routing controls, even with collapsed terminal output; all three uniquely tagged calculations complete in native child sessions; only the first and third show the subagent-routing banner and produce live gateway @@ -336,7 +356,13 @@ def test_smart_router_skill_toggles_claude_subagent_routing( """ session = live_session session.env["TMPDIR"] = str(tmp_path) - session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION + if SMART_ROUTER_CONFIG_VERSION is None: + session.env["ENABLE_SMART_ROUTING_V2"] = "1" + session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" + if orchestration_enabled: + session.env["ENABLE_SMART_ROUTER_ORCHESTRATOR"] = "1" + else: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION use_managed_config_fixture(session, "claude_smart_routing") session.run( "configure", @@ -356,7 +382,7 @@ def test_smart_router_skill_toggles_claude_subagent_routing( session, "claude", enabled=False, - SMART_ROUTER_CONFIG_VERSION=SMART_ROUTER_CONFIG_VERSION, + orchestration_enabled=orchestration_enabled, ) _run_calculation(tui, session, "claude", "1+2", "3", routed=False) _toggle_with_skill( @@ -364,7 +390,7 @@ def test_smart_router_skill_toggles_claude_subagent_routing( session, "claude", enabled=True, - SMART_ROUTER_CONFIG_VERSION=SMART_ROUTER_CONFIG_VERSION, + orchestration_enabled=orchestration_enabled, ) _run_calculation(tui, session, "claude", "2+2", "4", routed=True) tui.exit_normally() @@ -379,19 +405,31 @@ def test_smart_router_skill_toggles_claude_subagent_routing( @pytest.mark.codex @pytest.mark.managed_fixture @pytest.mark.parametrize( - "SMART_ROUTER_CONFIG_VERSION", - ["subagent_only_v0", "subagent_only_v1", "subagent_orch_v0"], - ids=["subagent_only_v0", "subagent_only_v1", "subagent_orch_v0"], + "orchestration_enabled, SMART_ROUTER_CONFIG_VERSION", + [ + (False, None), + (True, None), + (False, "subagent_only_v0"), + (False, "subagent_only_v1"), + (True, "subagent_orch_v0"), + ], + ids=[ + "routing-only", + "orchestration", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], ) def test_smart_router_skill_toggles_codex_subagent_routing( - live_session, workspace, tmp_path, SMART_ROUTER_CONFIG_VERSION + live_session, workspace, tmp_path, orchestration_enabled, SMART_ROUTER_CONFIG_VERSION ): - """Scenario: launch Codex with each supported subagent selector, spawn a child, invoke + """Scenario: launch Codex with the legacy flags or a subagent selector, spawn a child, invoke the installed Smart Router skill to turn routing off, spawn another child, turn routing back on through the skill, and spawn a third child in the same real TUI session. - Expected: subagent_only_v0 and subagent_only_v1 install Smart Router, while - subagent_orch_v0 also installs Smart Router Orchestrator. Each invocation records the CLI + Expected: routing-only cases install Smart Router; enabling orchestration also installs + Smart Router Orchestrator. Each invocation records the CLI confirmation in the native transcript and changes the saved routing controls, even with collapsed terminal output; all three uniquely tagged calculations complete in native child sessions; only the first and third show the subagent-routing banner and produce live gateway @@ -399,7 +437,13 @@ def test_smart_router_skill_toggles_codex_subagent_routing( """ session = live_session session.env["TMPDIR"] = str(tmp_path) - session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION + if SMART_ROUTER_CONFIG_VERSION is None: + session.env["ENABLE_SMART_ROUTING_V2"] = "1" + session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" + if orchestration_enabled: + session.env["ENABLE_SMART_ROUTER_ORCHESTRATOR"] = "1" + else: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION use_managed_config_fixture(session, "codex_smart_routing") session.run( "configure", @@ -419,7 +463,7 @@ def test_smart_router_skill_toggles_codex_subagent_routing( session, "codex", enabled=False, - SMART_ROUTER_CONFIG_VERSION=SMART_ROUTER_CONFIG_VERSION, + orchestration_enabled=orchestration_enabled, ) _run_calculation(tui, session, "codex", "1+2", "3", routed=False) _toggle_with_skill( @@ -427,7 +471,7 @@ def test_smart_router_skill_toggles_codex_subagent_routing( session, "codex", enabled=True, - SMART_ROUTER_CONFIG_VERSION=SMART_ROUTER_CONFIG_VERSION, + orchestration_enabled=orchestration_enabled, ) _run_calculation(tui, session, "codex", "2+2", "4", routed=True) tui.exit_normally() From e84d777f4dc963c29cc062b48073cd97ce1fe849 Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Thu, 8 Oct 2026 23:28:32 +0000 Subject: [PATCH 13/30] Cover four smart router presets in the routing CUJ --- tests/README.md | 7 +- tests/e2e_cuj/README.md | 35 ++++++ tests/e2e_cuj/test_cuj_smart_routing.py | 135 +++++++++++++++++++++++- tests/integration/README.md | 6 ++ 4 files changed, 180 insertions(+), 3 deletions(-) diff --git a/tests/README.md b/tests/README.md index 1adfdc327..79e9a5aea 100644 --- a/tests/README.md +++ b/tests/README.md @@ -22,8 +22,11 @@ CUJs never republish configuration or create a remote reservation. CUJ helper tests also verify that unsupported agent names fail rather than defaulting to Codex. They cover Claude/Codex helper dispatch and rejection of routing decisions without the agent-specific prompt-submission evidence. -The smart-routing CUJ runs four fresh sessions: routed and explicit model for both -Claude and Codex. Routing-disabled coverage is deferred until a separately +The original smart-routing CUJ runs four fresh sessions: routed and explicit model for both +Claude and Codex. One additional test has four `SMART_ROUTER_CONFIG_VERSION` cases. +Each case exercises both agents and checks first-prompt routing, orchestrator context in +inference input, and a completed explicitly requested routed subagent. It does not establish +automatic orchestrator delegation. Routing-disabled coverage is deferred until a separately preconfigured workspace is assigned. `test_entry_points.py` also runs both installed console scripts (`ug` and `ucode`) diff --git a/tests/e2e_cuj/README.md b/tests/e2e_cuj/README.md index 52f5a9fbb..4c683f758 100644 --- a/tests/e2e_cuj/README.md +++ b/tests/e2e_cuj/README.md @@ -55,3 +55,38 @@ Collection only (no authentication or inference): uv run pytest -c tests/e2e_cuj/pytest.ini --confcutdir=tests/e2e_cuj \ --collect-only tests/e2e_cuj ``` + +## Smart-routing CUJ + +`test_cuj_smart_routing.py` uses its own read-only workspace with managed smart routing enabled. +The original tests remain unchanged. One additional test has four +`SMART_ROUTER_CONFIG_VERSION` cases; each launches Claude and Codex once. +The independent expectation table is: + +| Selector | First prompt routed | Orchestrator context | Child on first prompt | Explicitly requested child | +| --- | --- | --- | --- | --- | +| `first_prompt_and_subagent_no_orch_v0` | Yes | No | No | Yes, routed | +| `subagent_only_v0` | No | No | No | Yes, routed | +| `subagent_only_v1` | No | No | No | Yes, routed | +| `subagent_orch_v0` | No | Yes | No | Yes, routed | + +First prompts explicitly forbid delegation to isolate first-prompt routing. Assertions require +the expected presence/absence of a prompt-correlated router request, successful inference on +the routed or configured default model, completed native file-task evidence, and no child session. +Orchestrator presence means its activation context reached the real gateway inference input, +not that an assistant echoed it or a skill merely existed on disk. +Each preset session then explicitly requests one child for a separate hidden-value +file task. Assertions require a native child transcript containing the value, the completed +parent answer, a correlated spawn-routing decision, and successful child inference on the +router's selected model. This tests requested delegation, not automatic orchestrator delegation. +Selector cases have separate TUI artifact names, and the session environment is restored afterward. +The existing Claude explicit-model precedence case remains skipped; the routing-disabled case +still requires a separately preconfigured workspace. + +With the same live prerequisites: + +```bash +uv run --with pexpect==4.9.0 --with pyte==0.8.2 pytest \ + --confcutdir=tests/e2e_cuj tests/e2e_cuj/test_cuj_smart_routing.py \ + -k test_smart_router_config_version -v +``` diff --git a/tests/e2e_cuj/test_cuj_smart_routing.py b/tests/e2e_cuj/test_cuj_smart_routing.py index 85936f076..7cfe6bd9c 100644 --- a/tests/e2e_cuj/test_cuj_smart_routing.py +++ b/tests/e2e_cuj/test_cuj_smart_routing.py @@ -4,7 +4,14 @@ import pytest -from tests.integration.utils.evidence import FileTask +from tests.integration.utils.evidence import ( + FileTask, + agent_sessions, + assert_subagent_routed, + assistant_answers, + is_child_session, + read_jsonl, +) from .base import BaseCujTest from .helpers.constants import CLAUDE, CODEX, INFERENCE_PATHS, CodingAgent @@ -14,6 +21,7 @@ ROUTING_PATH = "/ai-gateway/routing/v1/routes:select" AGENTS = (CLAUDE, CODEX) +ORCHESTRATOR_CONTEXT = "Smart Router Orchestrator is on for this session." @dataclass(frozen=True) @@ -166,6 +174,131 @@ def completed_smart_routing_runs(cuj): class TestCujSmartRouting(BaseCujTest): WORKSPACE_URL = "https://dbc-1a9622fc-2e91.cloud.databricks.com/" + @pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION, first_prompt_routed, orchestrator_enabled", + [ + ("first_prompt_and_subagent_no_orch_v0", True, False), + ("subagent_only_v0", False, False), + ("subagent_only_v1", False, False), + ("subagent_orch_v0", False, True), + ], + ) + def test_smart_router_config_version( + self, cuj, SMART_ROUTER_CONFIG_VERSION, first_prompt_routed, orchestrator_enabled + ): + """Scenario: launch both agents with a preset, then explicitly request a subagent. + + Expected: first-prompt routing and orchestrator context match the preset; + one routed native child completes the delegated task. + """ + session, workspace, recorder = cuj + previous = session.env.get("SMART_ROUTER_CONFIG_VERSION") + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION + try: + configs = _assert_published_config_matches_expectations(workspace.config()) + recorder.configure_session(session, ["configure", "--disable-databricks-ai-tools"]) + for agent in AGENTS: + supported = workspace.model_ids(agent) + evidence = SessionEvidence(session.home, agent) + existing_sessions = set(agent_sessions(session, agent)) + task = FileTask(session) + task.prompt += " Do not delegate." + checkpoint = recorder.checkpoint() + recorder.prepare_launch() + with Terminal(session, f"{SMART_ROUTER_CONFIG_VERSION}-{agent}", [agent]) as tui: + tui.boot(timeout=150) + tui.submit(task.prompt) + tui.task(evidence, task) + requests = recorder.requests_after(checkpoint) + routes = [request for request in requests if request.path == ROUTING_PATH] + assert len(routes) == int(first_prompt_routed), (agent, routes) + expected_model = configs[agent]["default_models"]["default_model"] + if first_prompt_routed: + assert routes[0].payload["task"]["prompt"] == task.prompt + response = recorder.response_for(routes[0]) + assert response.status_code == 200 + selections = response.payload["route_selection"] + assert len(selections) == 1 + expected_model = selections[0]["route_option"]["model"] + inference = next( + request + for request in requests + if request.method == "POST" and request.path == INFERENCE_PATHS[agent] + ) + assert recorder.response_for(inference).status_code == 200 + assert canonical_model(inference.payload["model"]) == canonical_model( + expected_model + ) + evidence.assert_applied(task, supported, expected=expected_model) + assert ( + ORCHESTRATOR_CONTEXT.encode() in inference.body + ) == orchestrator_enabled, agent + assert not any( + is_child_session(agent, path, records) + for path, records in agent_sessions(session, agent).items() + if path not in existing_sessions + ), agent + + child_task = FileTask(session) + child_task.prompt = child_task.delegate_prompt + decisions_path = ( + session.home / ".ucode" / f"{agent}-smart-routing-decisions.jsonl" + ) + decision_count = len(read_jsonl(decisions_path)) + checkpoint = recorder.checkpoint() + tui.submit(child_task.prompt) + tui.task(evidence, child_task) + tui.wait_for( + lambda screen, task=child_task, agent=agent: task.completed( + session, agent, child=True + ), + "completed native subagent file task", + timeout=240, + ) + children = { + path: records + for path, records in agent_sessions(session, agent).items() + if path not in existing_sessions and is_child_session(agent, path, records) + } + assert len(children) == 1, (agent, children.keys()) + assert any( + child_task.value in answer + for records in children.values() + for answer in assistant_answers(agent, records) + ), agent + decisions = read_jsonl(decisions_path)[decision_count:] + assert_subagent_routed( + session, + agent, + child_task, + decision_ids={decision["decision_id"] for decision in decisions}, + ) + requests = recorder.requests_after(checkpoint) + routes = [request for request in requests if request.path == ROUTING_PATH] + assert len(routes) == 1, (agent, routes) + assert child_task.filename in routes[0].payload["task"]["prompt"] + response = recorder.response_for(routes[0]) + assert response.status_code == 200 + selections = response.payload["route_selection"] + assert len(selections) == 1 + inference = next( + request + for request in requests + if request.method == "POST" + and request.path == INFERENCE_PATHS[agent] + and request.sequence > routes[0].sequence + ) + assert recorder.response_for(inference).status_code == 200 + assert canonical_model(inference.payload["model"]) == canonical_model( + selections[0]["route_option"]["model"] + ) + tui.exit_normally() + finally: + if previous is None: + session.env.pop("SMART_ROUTER_CONFIG_VERSION", None) + else: + session.env["SMART_ROUTER_CONFIG_VERSION"] = previous + @pytest.mark.parametrize("agent", AGENTS) def test_agent_completes_real_first_prompt_file_task_without_model_override( self, completed_smart_routing_runs, agent diff --git a/tests/integration/README.md b/tests/integration/README.md index 7716dcc08..a5a1d48c4 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -9,6 +9,12 @@ is read-only and checked for changes at teardown; concurrent readers need no res CUJ2 adds three separately collected cases for exact MPS/MCP configuration, Codex inference, and Claude inference. +The dedicated smart-routing CUJ in `../e2e_cuj/test_cuj_smart_routing.py` leaves the original +managed-default and explicit-model cases unchanged. One additional test runs the four supported +`SMART_ROUTER_CONFIG_VERSION` values, checking both agents' first-prompt routing, orchestrator +context in inference input, and completed explicitly requested routed subagents. +This is not automatic orchestrator-delegation coverage. No workspace configuration is modified. + The [catalog discovery journey](../e2e_cuj/README.md) uses the CUJ3 workspace to check agent-compatible pickers, schema exclusions, configured defaults, and real inference. CI collects it through the shared `dedicated-cuj` job. From 142994ee9cecd974eaa4fe223fff298c746dcf96 Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Thu, 8 Oct 2026 23:37:37 +0000 Subject: [PATCH 14/30] Name and document smart router configuration versions --- src/ucode/smart_routing/config.py | 20 ++++++++++++++++---- tests/README.md | 3 ++- tests/test_smart_routing_config.py | 16 ++++++++-------- 3 files changed, 26 insertions(+), 13 deletions(-) diff --git a/src/ucode/smart_routing/config.py b/src/ucode/smart_routing/config.py index d64de93f7..89b1bcdd8 100644 --- a/src/ucode/smart_routing/config.py +++ b/src/ucode/smart_routing/config.py @@ -13,23 +13,35 @@ SMART_ROUTING_ENV_KEYS, ) +# Customer preset: route the first prompt and subagents, without orchestration. +FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0 = "first_prompt_and_subagent_no_orch_v0" + +# Route only subagents, with V2 disabled and no orchestration. +SUBAGENT_ONLY_V0 = "subagent_only_v0" + +# Route only subagents, with both V2 and subagent-only flags enabled; no orchestration. +SUBAGENT_ONLY_V1 = "subagent_only_v1" + +# Route only subagents and inject the Smart Router Orchestrator workflow. +SUBAGENT_ORCH_V0 = "subagent_orch_v0" + _VERSIONS = { - "first_prompt_and_subagent_no_orch_v0": { + FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0: { ENABLE_SMART_ROUTING_ENV_VAR: "1", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", }, - "subagent_only_v0": { + SUBAGENT_ONLY_V0: { ENABLE_SMART_ROUTING_ENV_VAR: "0", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", }, - "subagent_only_v1": { + SUBAGENT_ONLY_V1: { ENABLE_SMART_ROUTING_ENV_VAR: "1", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", }, - "subagent_orch_v0": { + SUBAGENT_ORCH_V0: { ENABLE_SMART_ROUTING_ENV_VAR: "0", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", diff --git a/tests/README.md b/tests/README.md index 011931a84..1b17864e6 100644 --- a/tests/README.md +++ b/tests/README.md @@ -143,7 +143,8 @@ PowerShell execution. `test_smart_routing_config.py` is a 135-case Cartesian component oracle: all three legacy routing flags take unset, `0`, and `1`, while the selector takes unset or one of the four -supported presets, including the customer first-prompt-and-subagent mode. It independently hardcodes each preset and asserts exact +supported presets, including the customer first-prompt-and-subagent mode. It uses the named +version constants but independently hardcodes each preset's settings and asserts exact `resolve_environment` and `apply_config` settings, true-unset omission, unrelated-key and input preservation, and valid-selector consumption. This file intentionally does not cover blank, whitespace-padded, unsuffixed, or unsupported selectors; import-time schema validation, diff --git a/tests/test_smart_routing_config.py b/tests/test_smart_routing_config.py index d7a15e941..508fe5d1a 100644 --- a/tests/test_smart_routing_config.py +++ b/tests/test_smart_routing_config.py @@ -13,22 +13,22 @@ from ucode.smart_routing import config _EXPECTED_PRESETS = { - "first_prompt_and_subagent_no_orch_v0": { + config.FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0: { ENABLE_SMART_ROUTING_ENV_VAR: "1", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", }, - "subagent_only_v0": { + config.SUBAGENT_ONLY_V0: { ENABLE_SMART_ROUTING_ENV_VAR: "0", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", }, - "subagent_only_v1": { + config.SUBAGENT_ONLY_V1: { ENABLE_SMART_ROUTING_ENV_VAR: "1", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", }, - "subagent_orch_v0": { + config.SUBAGENT_ORCH_V0: { ENABLE_SMART_ROUTING_ENV_VAR: "0", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", @@ -43,10 +43,10 @@ "SMART_ROUTER_CONFIG_VERSION", [ None, - "first_prompt_and_subagent_no_orch_v0", - "subagent_only_v0", - "subagent_only_v1", - "subagent_orch_v0", + config.FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0, + config.SUBAGENT_ONLY_V0, + config.SUBAGENT_ONLY_V1, + config.SUBAGENT_ORCH_V0, ], ) def test_smart_routing_config_cartesian_grid( From 7d36e28d3c0c8876cdd2c2a83c473cc90226a13b Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Thu, 8 Oct 2026 23:46:22 +0000 Subject: [PATCH 15/30] Use named smart router versions in the routing CUJ --- tests/e2e_cuj/README.md | 7 ++++--- tests/e2e_cuj/test_cuj4_smart_routing.py | 14 ++++++++++---- 2 files changed, 14 insertions(+), 7 deletions(-) diff --git a/tests/e2e_cuj/README.md b/tests/e2e_cuj/README.md index bd2e7f78d..2b5b75423 100644 --- a/tests/e2e_cuj/README.md +++ b/tests/e2e_cuj/README.md @@ -57,9 +57,10 @@ uv run pytest -c tests/e2e_cuj/pytest.ini --confcutdir=tests/e2e_cuj \ ## Smart-routing CUJ -`test_cuj_smart_routing.py` uses its own read-only workspace with managed smart routing enabled. +`test_cuj4_smart_routing.py` uses its own read-only workspace with managed smart routing enabled. The original tests remain unchanged. One additional test has four -`SMART_ROUTER_CONFIG_VERSION` cases; each launches Claude and Codex once. +`SMART_ROUTER_CONFIG_VERSION` cases using the shared version constants; each launches Claude +and Codex once. The independent expectation table is: | Selector | First prompt routed | Orchestrator context | Child on first prompt | Explicitly requested child | @@ -86,6 +87,6 @@ With the same live prerequisites: ```bash uv run --with pexpect==4.9.0 --with pyte==0.8.2 pytest \ - --confcutdir=tests/e2e_cuj tests/e2e_cuj/test_cuj_smart_routing.py \ + --confcutdir=tests/e2e_cuj tests/e2e_cuj/test_cuj4_smart_routing.py \ -k test_smart_router_config_version -v ``` diff --git a/tests/e2e_cuj/test_cuj4_smart_routing.py b/tests/e2e_cuj/test_cuj4_smart_routing.py index 69127dede..1a7080d99 100644 --- a/tests/e2e_cuj/test_cuj4_smart_routing.py +++ b/tests/e2e_cuj/test_cuj4_smart_routing.py @@ -12,6 +12,12 @@ is_child_session, read_jsonl, ) +from ucode.smart_routing.config import ( + FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0, + SUBAGENT_ONLY_V0, + SUBAGENT_ONLY_V1, + SUBAGENT_ORCH_V0, +) from .base import BaseCujTest from .helpers.constants import CLAUDE, CODEX, INFERENCE_PATHS, CodingAgent @@ -189,10 +195,10 @@ class TestCujSmartRouting(BaseCujTest): @pytest.mark.parametrize( "SMART_ROUTER_CONFIG_VERSION, first_prompt_routed, orchestrator_enabled", [ - ("first_prompt_and_subagent_no_orch_v0", True, False), - ("subagent_only_v0", False, False), - ("subagent_only_v1", False, False), - ("subagent_orch_v0", False, True), + (FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0, True, False), + (SUBAGENT_ONLY_V0, False, False), + (SUBAGENT_ONLY_V1, False, False), + (SUBAGENT_ORCH_V0, False, True), ], ) def test_smart_router_config_version( From 0c82331316dac2e7895dad91811c4c2909f1c62a Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Fri, 9 Oct 2026 00:10:03 +0000 Subject: [PATCH 16/30] Fix async Claude CUJ evidence and clean up routing presets --- tests/README.md | 3 + tests/e2e_cuj/README.md | 2 + tests/e2e_cuj/helpers/evidence.py | 6 +- tests/e2e_cuj/test_cuj4_smart_routing.py | 4 ++ tests/integration/README.md | 3 + tests/test_cuj_evidence.py | 78 ++++++++++++++++++++++++ 6 files changed, 95 insertions(+), 1 deletion(-) diff --git a/tests/README.md b/tests/README.md index 1b17864e6..0ecd91473 100644 --- a/tests/README.md +++ b/tests/README.md @@ -22,6 +22,9 @@ CUJs never republish configuration or create a remote reservation. CUJ helper tests also verify that unsupported agent names fail rather than defaulting to Codex. They cover Claude/Codex helper dispatch and rejection of routing decisions without the agent-specific prompt-submission evidence. +Native evidence-reader regressions cover Claude's background-agent completion notifications; +notifications alone cannot substitute for a final parent answer. These offline checks do not +establish a live routing pass. The original smart-routing CUJ runs four fresh sessions: routed and explicit model for both Claude and Codex. One additional test has four `SMART_ROUTER_CONFIG_VERSION` cases. Each case exercises both agents and checks first-prompt routing, orchestrator context in diff --git a/tests/e2e_cuj/README.md b/tests/e2e_cuj/README.md index 2b5b75423..4d33b66c0 100644 --- a/tests/e2e_cuj/README.md +++ b/tests/e2e_cuj/README.md @@ -80,6 +80,8 @@ file task. Assertions require a native child transcript containing the value, th parent answer, a correlated spawn-routing decision, and successful child inference on the router's selected model. This tests requested delegation, not automatic orchestrator delegation. Selector cases have separate TUI artifact names, and the session environment is restored afterward. +Each preset also runs public `ug revert` in cleanup, including after a failed assertion, so +interactive launches' OS-managed settings cannot contaminate the next preset's configuration. The existing Claude explicit-model precedence case remains skipped; the routing-disabled case still requires a separately preconfigured workspace. diff --git a/tests/e2e_cuj/helpers/evidence.py b/tests/e2e_cuj/helpers/evidence.py index eb75da959..877a54871 100644 --- a/tests/e2e_cuj/helpers/evidence.py +++ b/tests/e2e_cuj/helpers/evidence.py @@ -122,7 +122,11 @@ def _completed_turn(records, task): responses = [] for row in records[prompts[0] + 1 :]: msg = row.get("message", {}) - if row.get("type") == "user" and isinstance(msg.get("content"), str): + if ( + row.get("type") == "user" + and isinstance(msg.get("content"), str) + and row.get("origin", {}).get("kind") != "task-notification" + ): break if row.get("type") == "assistant": assert row.get("sessionId") == session_id and session_id diff --git a/tests/e2e_cuj/test_cuj4_smart_routing.py b/tests/e2e_cuj/test_cuj4_smart_routing.py index 1a7080d99..b705128b1 100644 --- a/tests/e2e_cuj/test_cuj4_smart_routing.py +++ b/tests/e2e_cuj/test_cuj4_smart_routing.py @@ -316,6 +316,10 @@ def test_smart_router_config_version( session.env.pop("SMART_ROUTER_CONFIG_VERSION", None) else: session.env["SMART_ROUTER_CONFIG_VERSION"] = previous + session.revert_machine_wide( + f"{SMART_ROUTER_CONFIG_VERSION}-revert", + "Smart-router preset cleanup left machine-wide agent settings", + ) @pytest.mark.parametrize("agent", AGENTS) def test_agent_completes_real_first_prompt_file_task_without_model_override( diff --git a/tests/integration/README.md b/tests/integration/README.md index 813d4d312..366e78bc6 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -15,6 +15,9 @@ managed-default and explicit-model cases unchanged. One additional test runs the `SMART_ROUTER_CONFIG_VERSION` values, checking both agents' first-prompt routing, orchestrator context in inference input, and completed explicitly requested routed subagents. This is not automatic orchestrator-delegation coverage. No workspace configuration is modified. +Each preset cleans up interactive OS-managed settings with public `ug revert`, even on failure. +Offline transcript tests check native Claude background-agent completion evidence; +a completion notification alone is not treated as the parent's completed answer. The [catalog discovery journey](../e2e_cuj/README.md) uses the CUJ3 workspace to check agent-compatible pickers, schema exclusions, configured defaults, and real inference. diff --git a/tests/test_cuj_evidence.py b/tests/test_cuj_evidence.py index fa50ef2ef..b2b632451 100644 --- a/tests/test_cuj_evidence.py +++ b/tests/test_cuj_evidence.py @@ -101,6 +101,84 @@ def test_cuj_evidence_completed_native_turn(tmp_path, agent): assert result["selected_model"] == model +def test_cuj_evidence_claude_async_notification_keeps_parent_turn(): + task = SimpleNamespace( + prompt="Delegate reading the file to one subagent.", value="hidden-value" + ) + prompt, answer = records(CLAUDE, task, "system.ai.claude-haiku-4-5") + delegation = copy.deepcopy(answer) + delegation["message"].update( + id="delegation", + stop_reason="tool_use", + content=[ + {"type": "tool_use", "id": "tool", "name": "Agent", "input": {"prompt": task.prompt}} + ], + ) + waiting = copy.deepcopy(answer) + waiting["message"].update( + id="waiting", content=[{"type": "text", "text": "Waiting for the child."}] + ) + rows = [ + prompt, + delegation, + { + "type": "user", + "sessionId": "session", + "message": { + "content": [{"type": "tool_result", "tool_use_id": "tool", "content": task.value}] + }, + }, + waiting, + { + "type": "user", + "sessionId": "session", + "origin": {"kind": "task-notification"}, + "promptSource": "system", + "turnOrigin": "task_notification", + "message": { + "content": ( + "\ncompleted\n" + f"{task.value}\n" + ) + }, + }, + answer, + ] + assert completed_turn(CLAUDE, rows[:3], task) is None + assert completed_turn(CLAUDE, rows[:-1], task) is None + turn = completed_turn(CLAUDE, rows, task) + assert turn is not None + assert (turn.session_id, turn.turn_id, turn.answer) == ("session", "response", task.value) + assert turn.models == ["system.ai.claude-haiku-4-5"] * 3 + child = copy.deepcopy(rows) + child[-1]["isSidechain"] = True + assert completed_turn(CLAUDE, child, task) is None + with pytest.raises(AssertionError, match="Prompt was submitted more than once"): + completed_turn(CLAUDE, [*rows, prompt], task) + missing_model = copy.deepcopy(rows) + del missing_model[3]["message"]["model"] + with pytest.raises(AssertionError, match="Missing inference model metadata"): + completed_turn(CLAUDE, missing_model, task) + + +def test_cuj_evidence_claude_real_user_message_ends_parent_turn(): + task = SimpleNamespace( + prompt="Delegate reading the file to one subagent.", value="hidden-value" + ) + prompt, answer = records(CLAUDE, task, "system.ai.claude-haiku-4-5") + user = { + "type": "user", + "sessionId": "session", + "origin": {"kind": "human"}, + "message": {"content": "A different task."}, + } + assert completed_turn(CLAUDE, [prompt, user, answer], task) is None + user["message"]["content"] = ( + f"{task.value}" + ) + assert completed_turn(CLAUDE, [prompt, user, answer], task) is None + + @pytest.mark.parametrize("agent", [CLAUDE, CODEX]) @pytest.mark.parametrize( "failure", From 0e5885412bfa7da9a32b0c0a48ad8c11f89bb627 Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Fri, 9 Oct 2026 01:12:14 +0000 Subject: [PATCH 17/30] Simplify routing presets and fix CUJ evidence matching --- AGENTS.md | 7 +- README.md | 29 ++- skills/smart-router-orchestrator/README.md | 6 +- src/ucode/cli.py | 19 +- src/ucode/smart_routing/config.py | 23 +- tests/README.md | 22 +- tests/e2e_cuj/README.md | 9 +- tests/e2e_cuj/helpers/evidence.py | 126 +++++++--- tests/e2e_cuj/test_cuj4_smart_routing.py | 125 ++++++++-- tests/integration/README.md | 22 +- tests/test_cli.py | 48 ++++ tests/test_cuj_evidence.py | 272 ++++++++++++++++++++- tests/test_smart_routing_config.py | 171 ++++++++++++- 13 files changed, 785 insertions(+), 94 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 66c35a81b..ed8b06ac2 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -37,9 +37,10 @@ are registered in `SMART_ROUTING_ENV_KEYS` in `src/ucode/constants.py`. An unset or empty selector preserves legacy environment-flag behavior. A valid nonempty selector overrides every conflicting legacy value in that registry. Do not use `setdefault` or preserve inherited values for version-owned parameters. -Resolve and materialize the selector at the CLI boundary before argument parsing or callbacks, -including setup, session controls, authentication, and managed-config discovery. Invalid -selectors must fail before those operations. Restore the inherited environment on every exit. +Resolve and materialize valid selectors at the CLI boundary before argument parsing or callbacks. +Unknown selectors are a no-op: do not apply a preset, mutate the inherited environment, or raise +an error. Existing legacy flags and explicit launch/session controls retain their behavior. +Restore the inherited environment on every exit. Explicit launch/session on/off controls still apply after version expansion. Managed routing defaults must not rewrite already-resolved version flags. diff --git a/README.md b/README.md index 1c877db49..bf5b5c2d2 100644 --- a/README.md +++ b/README.md @@ -256,6 +256,7 @@ Use `SMART_ROUTER_CONFIG_VERSION` at launch to select a smart-routing configurat | `subagent_only_v0` | On | Off | Off | | `subagent_only_v1` | On | Off | Off | | `subagent_orch_v0` | On | Off | On | +| `subagent_orch_v1` | On | Off | On | `first_prompt_and_subagent_no_orch_v0` is the customer configuration for first-prompt and subagent routing without orchestration: `ENABLE_SMART_ROUTING_V2=1`, @@ -265,8 +266,30 @@ and subagent routing without orchestration: `ENABLE_SMART_ROUTING_V2=1`, `ENABLE_SMART_ROUTING_SUBAGENT_ONLY` to `"1"`. Subagent-only takes precedence, so first-prompt routing remains off; orchestration is also off. +`subagent_orch_v1` enables all three legacy flags. Like `subagent_only_v1`, it routes +subagents rather than the first prompt, and it additionally enables orchestration. + +The following pairs select equivalent routing settings. Set the variables on the same command +line (or export them); separate unexported assignments joined by `&&` do not reliably reach UG. +Use the same `SMART_ROUTER_NAME` on both sides to select the same router. + +```bash +ENABLE_SMART_ROUTING_V2=1 ENABLE_SMART_ROUTING_SUBAGENT_ONLY=1 SMART_ROUTER_NAME=m2-r315-quality-20260929 uv run ug claude +SMART_ROUTER_CONFIG_VERSION=subagent_only_v1 SMART_ROUTER_NAME=m2-r315-quality-20260929 uv run ug claude + +ENABLE_SMART_ROUTING_V2=1 ENABLE_SMART_ROUTING_SUBAGENT_ONLY=1 ENABLE_SMART_ROUTER_ORCHESTRATOR=1 SMART_ROUTER_NAME=m2-r315-quality-20260929 uv run ug claude +SMART_ROUTER_CONFIG_VERSION=subagent_orch_v1 SMART_ROUTER_NAME=m2-r315-quality-20260929 uv run ug claude + +ENABLE_SMART_ROUTING_V2=1 uv run ug claude +SMART_ROUTER_CONFIG_VERSION=first_prompt_and_subagent_no_orch_v0 uv run ug claude +``` + +These pairs assume other routing flags are unset or `"0"`. The version selector overrides +inherited routing flags; legacy assignments leave unspecified flags inherited. +The orchestration flag is `ENABLE_SMART_ROUTER_ORCHESTRATOR`, not `ENABLE_SMART_ROUTING_ORCH`. + Smart-routed Claude and Codex sessions install `smart-router`. The `subagent_orch_v0` -version also installs and activates the bundled `smart-router-orchestrator` skill. +and `subagent_orch_v1` versions also install and activate the bundled `smart-router-orchestrator` skill. For example: ```bash @@ -278,8 +301,8 @@ or running any command callbacks, UG expands it into `ENABLE_SMART_ROUTING_V2`, `ENABLE_SMART_ROUTING_SUBAGENT_ONLY`, and `ENABLE_SMART_ROUTER_ORCHESTRATOR` for the launched session. When the version is unset or empty, these legacy flags retain their existing behavior, including -first-prompt routing through `ENABLE_SMART_ROUTING_V2=1`. Unknown versions produce -an error listing the supported values before setup, authentication, or session changes. +first-prompt routing through `ENABLE_SMART_ROUTING_V2=1`. Unknown versions are ignored: +no preset is applied, the inherited environment is unchanged, and commands continue normally. Explicit launch/session on/off controls apply after expansion. Orchestration remains off by default. Workspace smart-routing defaults do not rewrite the selected version's flags. diff --git a/skills/smart-router-orchestrator/README.md b/skills/smart-router-orchestrator/README.md index 081be4f6e..c22625d0a 100644 --- a/skills/smart-router-orchestrator/README.md +++ b/skills/smart-router-orchestrator/README.md @@ -2,8 +2,10 @@ UG bundles the `smart-router-orchestrator` workflow and five Claude role definitions. Smart-routed Claude and Codex launches install and activate this skill alongside -`smart-router` with `SMART_ROUTER_CONFIG_VERSION=subagent_orch_v0`. -UG expands that version into the session's legacy feature flags. +`smart-router` with `SMART_ROUTER_CONFIG_VERSION=subagent_orch_v0` or `subagent_orch_v1`. +UG expands the selected version into the session's legacy feature flags. The `_v1` revision +enables both V2 and subagent-only routing flags alongside orchestration; subagent-only still +takes precedence, so the first prompt is not routed. The existing `ENABLE_SMART_ROUTER_ORCHESTRATOR=1` opt-in remains supported when `SMART_ROUTER_CONFIG_VERSION` is unset; `subagent_only_v0` explicitly leaves orchestration off. The feature is off by default; routing alone installs only `smart-router`. diff --git a/src/ucode/cli.py b/src/ucode/cli.py index 21fe69b66..7c1746ac7 100644 --- a/src/ucode/cli.py +++ b/src/ucode/cli.py @@ -1333,10 +1333,7 @@ def make_context( parent: _click.Context | None = None, **extra: Any, ) -> _click.Context: - try: - previous = smart_routing_v2.apply_config() - except RuntimeError as exc: - raise _click.ClickException(str(exc)) from None + previous = smart_routing_v2.apply_config() try: ctx = super().make_context(info_name, args, parent, **extra) except BaseException: @@ -2306,15 +2303,11 @@ def _auto_configure_tool(tool: str, custom_oauth: CustomOAuthConfig | None = Non @contextmanager def _smart_routing_v2_flag(enabled: bool | None) -> Iterator[None]: """Apply an explicit routing choice without leaking into an embedding process.""" - try: - previous = ( - smart_routing_v2.apply_config() - if enabled is None - else smart_routing_v2.override_smart_routing(enabled) - ) - except RuntimeError as exc: - print_err(str(exc)) - raise typer.Exit(1) from None + previous = ( + smart_routing_v2.apply_config() + if enabled is None + else smart_routing_v2.override_smart_routing(enabled) + ) try: yield finally: diff --git a/src/ucode/smart_routing/config.py b/src/ucode/smart_routing/config.py index 89b1bcdd8..2a7fe9c01 100644 --- a/src/ucode/smart_routing/config.py +++ b/src/ucode/smart_routing/config.py @@ -25,6 +25,9 @@ # Route only subagents and inject the Smart Router Orchestrator workflow. SUBAGENT_ORCH_V0 = "subagent_orch_v0" +# Route only subagents with V2, subagent-only, and orchestration all enabled. +SUBAGENT_ORCH_V1 = "subagent_orch_v1" + _VERSIONS = { FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0: { ENABLE_SMART_ROUTING_ENV_VAR: "1", @@ -46,6 +49,11 @@ ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", }, + SUBAGENT_ORCH_V1: { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + }, } @@ -75,14 +83,7 @@ def resolve_environment(env: Mapping[str, str] | None = None) -> dict[str, str]: """Expand a version before applying any launch or session-specific overrides.""" resolved = dict(os.environ if env is None else env) version = resolved.pop(SMART_ROUTER_CONFIG_VERSION_ENV_VAR, "").strip() - if not version: - return resolved - if version not in _VERSIONS: - raise RuntimeError( - f"Unknown {SMART_ROUTER_CONFIG_VERSION_ENV_VAR} value {version!r}. " - f"Use one of: {', '.join(_VERSIONS)}, or unset it to use the legacy flags." - ) - resolved.update(_VERSIONS[version]) + resolved.update(_VERSIONS.get(version, {})) return resolved @@ -90,11 +91,11 @@ def apply_config(env: MutableMapping[str, str] | None = None) -> dict[str, str | """Consume the launch selector, returning the values needed to restore its input.""" target = os.environ if env is None else env version = target.get(SMART_ROUTER_CONFIG_VERSION_ENV_VAR, "").strip() - if not version: + preset = _VERSIONS.get(version) + if preset is None: return {} - resolved = resolve_environment(target) keys = (*SMART_ROUTING_ENV_KEYS, SMART_ROUTER_CONFIG_VERSION_ENV_VAR) previous = {key: target.get(key) for key in keys} - target.update({key: resolved[key] for key in SMART_ROUTING_ENV_KEYS}) + target.update(preset) target.pop(SMART_ROUTER_CONFIG_VERSION_ENV_VAR, None) return previous diff --git a/tests/README.md b/tests/README.md index 0ecd91473..741d51d2b 100644 --- a/tests/README.md +++ b/tests/README.md @@ -25,8 +25,12 @@ the agent-specific prompt-submission evidence. Native evidence-reader regressions cover Claude's background-agent completion notifications; notifications alone cannot substitute for a final parent answer. These offline checks do not establish a live routing pass. +Offline CUJ regressions also cover Codex delegated-turn completion, including mirrored prompt +records and child notifications, while rejecting incomplete or unrelated turns. Smart-routing +request selection matches exact task prompts in tool-capable inference payloads, excluding +session-title requests and parent continuations when checking child inference. The original smart-routing CUJ runs four fresh sessions: routed and explicit model for both -Claude and Codex. One additional test has four `SMART_ROUTER_CONFIG_VERSION` cases. +Claude and Codex. One additional test has five `SMART_ROUTER_CONFIG_VERSION` cases. Each case exercises both agents and checks first-prompt routing, orchestrator context in inference input, and a completed explicitly requested routed subagent. It does not establish automatic orchestrator delegation. Routing-disabled coverage is deferred until a separately @@ -144,16 +148,22 @@ that Claude settings and Codex's shell policy carry the interpreter and session These are component checks; they do not establish native skill permission matching or PowerShell execution. -`test_smart_routing_config.py` is a 135-case Cartesian component oracle: all three legacy -routing flags take unset, `0`, and `1`, while the selector takes unset or one of the four +`test_smart_routing_config.py` includes a 162-case Cartesian component oracle: all three legacy +routing flags take unset, `0`, and `1`, while the selector takes unset or one of the five supported presets, including the customer first-prompt-and-subagent mode. It uses the named version constants but independently hardcodes each preset's settings and asserts exact `resolve_environment` and `apply_config` settings, true-unset omission, unrelated-key and -input preservation, and valid-selector consumption. This file intentionally does not cover -blank, whitespace-padded, unsuffixed, or unsupported selectors; import-time schema validation, -CLI startup ordering, snapshots/restoration, routing getters, native subcommands, hooks, +input preservation, and valid-selector consumption. Additional helper cases cover no-op behavior +for unsupported selectors, valid-selector application/restoration, and unset/empty selectors. +CLI regressions in `test_cli.py` verify that unknown selectors do not block token-only +authentication, revert, or Claude/Codex/default launch dispatch, and preserve the environment. +Import-time schema validation, +routing getters, native subcommands, hooks, session files, or managed launches are not established by this grid. These are component checks; they do not establish live agent, hook, or gateway behavior. +Three additional component cases compare the legacy flags with `subagent_only_v1`, +`subagent_orch_v1`, and `first_prompt_and_subagent_no_orch_v0`, including first-prompt, +subagent-routing activation, orchestration, and preservation of the chosen router name. The toggle integration journeys retain the legacy routing-only and orchestration cases and add the three subagent-only `SMART_ROUTER_CONFIG_VERSION` presets. They require only diff --git a/tests/e2e_cuj/README.md b/tests/e2e_cuj/README.md index 4d33b66c0..88e349957 100644 --- a/tests/e2e_cuj/README.md +++ b/tests/e2e_cuj/README.md @@ -58,7 +58,7 @@ uv run pytest -c tests/e2e_cuj/pytest.ini --confcutdir=tests/e2e_cuj \ ## Smart-routing CUJ `test_cuj4_smart_routing.py` uses its own read-only workspace with managed smart routing enabled. -The original tests remain unchanged. One additional test has four +The original tests remain unchanged. One additional test has five `SMART_ROUTER_CONFIG_VERSION` cases using the shared version constants; each launches Claude and Codex once. The independent expectation table is: @@ -69,16 +69,23 @@ The independent expectation table is: | `subagent_only_v0` | No | No | No | Yes, routed | | `subagent_only_v1` | No | No | No | Yes, routed | | `subagent_orch_v0` | No | Yes | No | Yes, routed | +| `subagent_orch_v1` | No | Yes | No | Yes, routed | First prompts explicitly forbid delegation to isolate first-prompt routing. Assertions require the expected presence/absence of a prompt-correlated router request, successful inference on the routed or configured default model, completed native file-task evidence, and no child session. Orchestrator presence means its activation context reached the real gateway inference input, not that an assistant echoed it or a skill merely existed on disk. +Task inference is selected by its exact user prompt and tool-capable payload, not the first +POST: session-title requests are excluded. Child requests use the routed prompt from the native +child transcript, so parent continuations cannot stand in for child inference. Each preset session then explicitly requests one child for a separate hidden-value file task. Assertions require a native child transcript containing the value, the completed parent answer, a correlated spawn-routing decision, and successful child inference on the router's selected model. This tests requested delegation, not automatic orchestrator delegation. +Codex completion evidence tolerates mirrored prompt records and child notifications but still +requires the matching native completed turn and final parent answer. Offline regressions cover +these transcript and request-selection cases; they do not establish a live CUJ pass. Selector cases have separate TUI artifact names, and the session environment is restored afterward. Each preset also runs public `ug revert` in cleanup, including after a failed assertion, so interactive launches' OS-managed settings cannot contaminate the next preset's configuration. diff --git a/tests/e2e_cuj/helpers/evidence.py b/tests/e2e_cuj/helpers/evidence.py index 877a54871..6db404ad7 100644 --- a/tests/e2e_cuj/helpers/evidence.py +++ b/tests/e2e_cuj/helpers/evidence.py @@ -154,48 +154,118 @@ def _completed_turn(records, task): meta = [row["payload"] for row in records if row.get("type") == "session_meta"] if len(meta) != 1 or isinstance(meta[0].get("source"), dict): return None - contexts, prompt_count, seen_prompt, active_turn = {}, 0, False, None + contexts, started, completed, aborted, current_turn = {}, set(), set(), set(), None + target_turn, prompt_sources, seen_prompt, final_answer = None, set(), False, None + + def notification_text(row): + payload = row.get("payload", {}) + if row.get("type") == "event_msg" and payload.get("type") == "user_message": + return payload.get("message", "") + if ( + row.get("type") == "response_item" + and payload.get("type") == "message" + and payload.get("role") == "user" + ): + return message_text(payload.get("content")) + return "" + + def is_notification(row): + text = notification_text(row) + normalized = text.lower() if isinstance(text, str) else "" + return ( + "" in normalized or "" in normalized + ) + + def row_turn(row): + payload = row.get("payload", {}) + if payload.get("turn_id"): + return payload["turn_id"] + metadata = row.get("internal_chat_message_metadata_passthrough") + return metadata.get("turn_id") if isinstance(metadata, dict) else None + + def handle_prompt(row, source): + nonlocal seen_prompt, target_turn + payload = row.get("payload", {}) + if row.get("type") == "response_item": + prompt = message_text(payload.get("content")) + else: + prompt = payload.get("message") + if prompt != task.prompt: + return False + if source in prompt_sources: + raise AssertionError("Prompt was submitted more than once") + prompt_sources.add(source) + if not seen_prompt: + seen_prompt = True + prompt_turn = row_turn(row) or current_turn + if target_turn is None and prompt_turn not in completed | aborted: + target_turn = prompt_turn + elif target_turn is not None and prompt_turn is not None: + assert prompt_turn == target_turn, "Prompt was submitted more than once" + return True + for row in records: payload = row.get("payload", {}) if row.get("type") == "turn_context": - contexts.setdefault(payload.get("turn_id"), []).append(payload.get("model")) + turn_id = payload.get("turn_id") + contexts.setdefault(turn_id, []).append(payload.get("model")) + current_turn = turn_id + if seen_prompt and target_turn is None and turn_id not in completed | aborted: + target_turn = turn_id + elif target_turn is not None and turn_id != target_turn: + return None if ( row.get("type") == "response_item" and payload.get("type") == "message" and payload.get("role") == "user" ): - prompt = message_text(payload.get("content")) - if prompt == task.prompt: - prompt_count += 1 - assert prompt_count == 1, "Prompt was submitted more than once" - seen_prompt = True - elif seen_prompt: + if handle_prompt(row, "response_item"): + continue + if seen_prompt and not is_notification(row): return None if row.get("type") != "event_msg": continue - if payload.get("type") == "task_started": - if seen_prompt: + event_type = payload.get("type") + if event_type == "task_started": + turn_id = payload.get("turn_id") + if not turn_id: + continue + started.add(turn_id) + current_turn = turn_id + if not seen_prompt: + continue + if target_turn is None and turn_id not in completed | aborted: + target_turn = turn_id + elif turn_id != target_turn: return None - active_turn = payload.get("turn_id") - if payload.get("type") == "user_message": - if payload.get("message") == task.prompt: - prompt_count += 1 - assert prompt_count == 1, "Prompt was submitted more than once" - seen_prompt = True - elif seen_prompt: + elif event_type == "user_message": + if handle_prompt(row, "user_message"): + continue + if seen_prompt and not is_notification(row): return None - if seen_prompt and payload.get("type") == "task_complete": + elif event_type == "task_complete": + turn_id = payload.get("turn_id") + completed.add(turn_id) + if seen_prompt and turn_id == target_turn: + final_answer = payload.get("last_agent_message") or "" + break + elif event_type == "turn_aborted": turn_id = payload.get("turn_id") - answer = payload.get("last_agent_message") or "" - if ( - task.value in answer - and turn_id - and turn_id == active_turn - and contexts.get(turn_id) - ): - assert all(contexts[turn_id]), "Missing native turn model metadata" - return CompletedTurn(meta[0]["id"], turn_id, contexts[turn_id], answer) - return None + aborted.add(turn_id) + if turn_id == target_turn: + return None + + if ( + not seen_prompt + or target_turn is None + or target_turn not in started + or final_answer is None + ): + return None + if task.value not in final_answer or not contexts.get(target_turn): + return None + assert all(contexts[target_turn]), "Missing native turn model metadata" + return CompletedTurn(meta[0]["id"], target_turn, contexts[target_turn], final_answer) def get_cuj_helper(agent): diff --git a/tests/e2e_cuj/test_cuj4_smart_routing.py b/tests/e2e_cuj/test_cuj4_smart_routing.py index b705128b1..ce86f6d20 100644 --- a/tests/e2e_cuj/test_cuj4_smart_routing.py +++ b/tests/e2e_cuj/test_cuj4_smart_routing.py @@ -17,6 +17,7 @@ SUBAGENT_ONLY_V0, SUBAGENT_ONLY_V1, SUBAGENT_ORCH_V0, + SUBAGENT_ORCH_V1, ) from .base import BaseCujTest @@ -26,6 +27,7 @@ SessionObservation, canonical_model, claude_file_task, + message_text, ) from .helpers.terminal import Terminal from .helpers.tui_request_recorder import RecordedRequest, RecordedResponse @@ -85,6 +87,77 @@ def _run_session(session, recorder, agent, task, launch_args): return evidence.observe(task), recorder.requests_after(checkpoint) +def _content_is_exact_prompt(content, prompt): + if isinstance(content, str): + return content == prompt + return isinstance(content, list) and any( + isinstance(part, dict) + and part.get("type") in {"text", "input_text"} + and part.get("text") == prompt + for part in content + ) + + +def _request_contains_exact_user_prompt(request, agent, prompt): + field = "messages" if agent == CLAUDE else "input" + entries = request.payload.get(field) + if isinstance(entries, str): + return entries == prompt + if not isinstance(entries, list): + return False + return any( + isinstance(entry, dict) + and entry.get("role") == "user" + and _content_is_exact_prompt(entry.get("content"), prompt) + for entry in entries + ) + + +def _is_tool_capable(request): + tools = request.payload.get("tools") + return isinstance(tools, list) and bool(tools) + + +def _task_inference_request(requests, agent, prompt, *, native_turn, after=0, native_prompts=()): + """Select a task request from payload evidence, never from its expected model.""" + assert native_turn is not None, "No completed native turn anchors the task request" + if native_prompts: + assert prompt in native_prompts, "Routed prompt is absent from the native child session" + matches = tuple( + request + for request in requests + if request.sequence > after + and request.method == "POST" + and request.path == INFERENCE_PATHS[agent] + and _request_contains_exact_user_prompt(request, agent, prompt) + and _is_tool_capable(request) + ) + assert matches, "No tool-capable inference request contained the exact task prompt" + return matches[0] + + +def _native_user_prompts(agent, records): + prompts = [] + for record in records: + if agent == CLAUDE and record.get("type") == "user": + prompt = message_text(record.get("message", {}).get("content")) + elif agent == CODEX and record.get("type") == "event_msg": + payload = record.get("payload", {}) + prompt = payload.get("message") if payload.get("type") == "user_message" else "" + elif agent == CODEX and record.get("type") == "response_item": + payload = record.get("payload", {}) + prompt = ( + message_text(payload.get("content")) + if payload.get("type") == "message" and payload.get("role") == "user" + else "" + ) + else: + prompt = "" + if isinstance(prompt, str) and prompt: + prompts.append(prompt) + return tuple(dict.fromkeys(prompts)) + + def _assert_published_config_matches_expectations(published): assert published["spec_version"] == 1 assert published["default_agent"] == CodingAgent.CLAUDE_CODE @@ -139,10 +212,11 @@ def run_smart_routing_journeys(cuj) -> SmartRoutingSessionResults: if request.method == "POST" and request.path == ROUTING_PATH ) route_response = recorder.response_for(route_request) - inference_request = next( - request - for request in requests - if request.method == "POST" and request.path == INFERENCE_PATHS[agent] + inference_request = _task_inference_request( + requests, + agent, + task.prompt, + native_turn=observation.turn, ) no_model_override[agent] = SessionCase( agent=agent, @@ -159,10 +233,11 @@ def run_smart_routing_journeys(cuj) -> SmartRoutingSessionResults: task = _file_task(session, agent) launch_args = (agent, "--model", overrides[agent]) observation, requests = _run_session(session, recorder, agent, task, launch_args) - inference_request = next( - request - for request in requests - if request.method == "POST" and request.path == INFERENCE_PATHS[agent] + inference_request = _task_inference_request( + requests, + agent, + task.prompt, + native_turn=observation.turn, ) with_model_override[agent] = SessionCase( agent=agent, @@ -199,6 +274,7 @@ class TestCujSmartRouting(BaseCujTest): (SUBAGENT_ONLY_V0, False, False), (SUBAGENT_ONLY_V1, False, False), (SUBAGENT_ORCH_V0, False, True), + (SUBAGENT_ORCH_V1, False, True), ], ) def test_smart_router_config_version( @@ -238,10 +314,12 @@ def test_smart_router_config_version( selections = response.payload["route_selection"] assert len(selections) == 1 expected_model = selections[0]["route_option"]["model"] - inference = next( - request - for request in requests - if request.method == "POST" and request.path == INFERENCE_PATHS[agent] + native_turn = evidence.completed(task) + inference = _task_inference_request( + requests, + agent, + task.prompt, + native_turn=native_turn, ) assert recorder.response_for(inference).status_code == 200 assert canonical_model(inference.payload["model"]) == canonical_model( @@ -284,6 +362,12 @@ def test_smart_router_config_version( for records in children.values() for answer in assistant_answers(agent, records) ), agent + native_turn = evidence.completed(child_task) + child_prompts = tuple( + prompt + for records in children.values() + for prompt in _native_user_prompts(agent, records) + ) decisions = read_jsonl(decisions_path)[decision_count:] assert_subagent_routed( session, @@ -294,17 +378,20 @@ def test_smart_router_config_version( requests = recorder.requests_after(checkpoint) routes = [request for request in requests if request.path == ROUTING_PATH] assert len(routes) == 1, (agent, routes) - assert child_task.filename in routes[0].payload["task"]["prompt"] + route_prompt = routes[0].payload["task"]["prompt"] + assert route_prompt in child_prompts, (agent, route_prompt, child_prompts) + assert child_task.filename in route_prompt response = recorder.response_for(routes[0]) assert response.status_code == 200 selections = response.payload["route_selection"] assert len(selections) == 1 - inference = next( - request - for request in requests - if request.method == "POST" - and request.path == INFERENCE_PATHS[agent] - and request.sequence > routes[0].sequence + inference = _task_inference_request( + requests, + agent, + route_prompt, + native_turn=native_turn, + after=routes[0].sequence, + native_prompts=child_prompts, ) assert recorder.response_for(inference).status_code == 200 assert canonical_model(inference.payload["model"]) == canonical_model( diff --git a/tests/integration/README.md b/tests/integration/README.md index 366e78bc6..0d57262ea 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -11,13 +11,17 @@ inference, and Claude inference. The config equality also accounts for the works skill name as data; skill download and invocation are not covered. The dedicated smart-routing CUJ in `../e2e_cuj/test_cuj_smart_routing.py` leaves the original -managed-default and explicit-model cases unchanged. One additional test runs the four supported +managed-default and explicit-model cases unchanged. One additional test runs the five supported `SMART_ROUTER_CONFIG_VERSION` values, checking both agents' first-prompt routing, orchestrator context in inference input, and completed explicitly requested routed subagents. This is not automatic orchestrator-delegation coverage. No workspace configuration is modified. Each preset cleans up interactive OS-managed settings with public `ug revert`, even on failure. Offline transcript tests check native Claude background-agent completion evidence; a completion notification alone is not treated as the parent's completed answer. +Offline CUJ cases additionally cover Codex delegated-turn completion and reject incomplete +or unrelated turns. Smart-routing requests are matched by exact task prompts and tool-capable +payloads rather than the first inference POST, excluding session-title and parent-continuation +traffic without weakening model, orchestrator-context, or paired-response assertions. The [catalog discovery journey](../e2e_cuj/README.md) uses the CUJ3 workspace to check agent-compatible pickers, schema exclusions, configured defaults, and real inference. @@ -319,15 +323,21 @@ their original legacy-env or managed-default cases remain. Explicit-model select forwarding, app-server initialization, and catalog fallback retain their original legacy flags and on/off coverage; preset-specific behavior there is not covered. -The unit/component `../test_smart_routing_config.py` is a 135-case Cartesian oracle over all -three legacy routing flags (`None`, `0`, `1`) and five selector forms (`None` plus the four +The unit/component `../test_smart_routing_config.py` includes a 162-case Cartesian oracle over all +three legacy routing flags (`None`, `0`, `1`) and six selector forms (`None` plus the five supported presets, including the customer first-prompt-and-subagent mode). It independently hardcodes preset values and asserts exact `resolve_environment` and `apply_config` settings, true-unset omission, unrelated-key and -input preservation, and valid-selector consumption. This file no longer covers blank, -whitespace-padded, unsuffixed, or unsupported selectors; import-time schema validation, CLI -startup ordering, snapshots/restoration, routing getters, native subcommands, hooks, session +input preservation, and valid-selector consumption. Additional helper cases cover no-op behavior +for unsupported selectors, valid-selector application/restoration, and unset/empty selectors. +Component cases in `../test_cli.py` check token-only authentication, revert, and +Claude/Codex/default launch dispatch with unknown selectors and environment preservation. +Import-time schema validation, +routing getters, native subcommands, hooks, session files, or managed launches are not established here. It does not claim live agent, hook, or gateway coverage. +Three component cases additionally compare the equivalent legacy flags and presets for +`subagent_only_v1`, `subagent_orch_v1`, and `first_prompt_and_subagent_no_orch_v0`; +they verify routing activation, first-prompt behavior, orchestration, and router-name preservation. The portable `../test_claude_windows_smart_routing.py` checks the Windows subagent-only fallback without Unix imports. Native Windows TUI and hook execution diff --git a/tests/test_cli.py b/tests/test_cli.py index a9d9a859c..f1d1e7d54 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -1898,6 +1898,21 @@ def test_host_and_profile_override_state(self): assert result.exit_code == 0 fetch.assert_called_once_with("https://override", "prod", force_refresh=False) + def test_invalid_routing_config_does_not_block_token(self, monkeypatch): + monkeypatch.setenv("SMART_ROUTER_CONFIG_VERSION", "unsupported_v99") + monkeypatch.setenv("ENABLE_SMART_ROUTING_V2", "1") + previous = dict(os.environ) + with ( + patch("ucode.cli.load_state", return_value={"workspace": "https://ws"}), + patch("ucode.cli.get_databricks_token", return_value="tok-123") as fetch, + ): + result = runner.invoke(app, ["auth-token"]) + + assert result.exit_code == 0, result.output + assert result.stdout == "tok-123\n" + fetch.assert_called_once_with("https://ws", None, force_refresh=False) + assert dict(os.environ) == previous + def test_force_refresh_is_forwarded(self): with ( patch("ucode.cli.load_state", return_value={"workspace": "https://ws"}), @@ -2844,6 +2859,17 @@ def test_skills_entry_absent_from_per_client_mcp_lines(self): class TestRevert: + def test_invalid_routing_config_does_not_block_revert(self, monkeypatch): + monkeypatch.setenv("SMART_ROUTER_CONFIG_VERSION", "unsupported_v99") + monkeypatch.setenv("ENABLE_SMART_ROUTING_V2", "1") + previous = dict(os.environ) + with patch("ucode.cli.revert", return_value=0) as revert: + result = runner.invoke(app, ["revert"]) + + assert result.exit_code == 0, result.output + revert.assert_called_once_with() + assert dict(os.environ) == previous + def test_reverts_mcp_configs_before_clearing_state(self): state = { **MINIMAL_STATE, @@ -2871,6 +2897,28 @@ def test_reverts_mcp_configs_before_clearing_state(self): assert "Claude Code MCP config: restored" in result.output +@pytest.mark.parametrize("args", [[], ["claude"], ["codex"]]) +def test_unknown_routing_config_does_not_block_launch(monkeypatch, args): + monkeypatch.setenv("SMART_ROUTER_CONFIG_VERSION", "unsupported_v99") + previous = dict(os.environ) + with ( + patch("ucode.cli._launch_tool") as launch, + patch("ucode.cli._launch_managed_default") as launch_default, + patch("ucode.cli.configure_shared_state") as configure, + ): + result = runner.invoke(app, args) + + assert result.exit_code == 0, result.output + if args: + launch.assert_called_once() + launch_default.assert_not_called() + else: + launch.assert_not_called() + launch_default.assert_called_once() + configure.assert_not_called() + assert dict(os.environ) == previous + + class TestDoctorCommand: def test_invokes_doctor(self): with patch("ucode.doctor.doctor", return_value=0) as mock_doctor: diff --git a/tests/test_cuj_evidence.py b/tests/test_cuj_evidence.py index b2b632451..a58644fbf 100644 --- a/tests/test_cuj_evidence.py +++ b/tests/test_cuj_evidence.py @@ -6,7 +6,7 @@ import pytest -from tests.e2e_cuj.helpers.constants import CLAUDE, CODEX +from tests.e2e_cuj.helpers.constants import CLAUDE, CODEX, INFERENCE_PATHS from tests.e2e_cuj.helpers.evidence import ( BaseCujHelper, ClaudeCujHelper, @@ -16,6 +16,7 @@ completed_turn, get_cuj_helper, ) +from tests.e2e_cuj.test_cuj4_smart_routing import _task_inference_request from tests.integration.utils.evidence import FileTask, read_jsonl @@ -227,6 +228,275 @@ def test_cuj_evidence_codex_rejects_wrong_turn(tmp_path): assert completed_turn(CODEX, wrong_turn, task) is None +def _codex_parent_delegation_records(initial_task, parent_task, model, child_marker): + rows = records(CODEX, initial_task, model) + rows.extend( + [ + { + "type": "event_msg", + "payload": {"type": "user_message", "message": parent_task.prompt}, + }, + {"type": "event_msg", "payload": {"type": "task_started", "turn_id": "delegate"}}, + { + "type": "turn_context", + "payload": {"turn_id": "delegate", "model": model}, + }, + { + "type": "response_item", + "payload": { + "type": "message", + "role": "user", + "content": [ + { + "type": "input_text", + "text": ( + "\n" + f'{{"agent_path":"child","status":{{"completed":"{child_marker}"}}}}\n' + "" + ), + } + ], + }, + "internal_chat_message_metadata_passthrough": { + "turn_id": "delegate", + "content_item_kinds": ["multi_agent.subagent_notification"], + }, + }, + { + "type": "event_msg", + "payload": { + "type": "task_complete", + "turn_id": "delegate", + "last_agent_message": parent_task.value, + }, + }, + ] + ) + return rows + + +def test_cuj_evidence_codex_parent_turn_survives_child_notification(tmp_path): + initial_task = FileTask(SimpleNamespace(cwd=tmp_path)) + parent_task = FileTask(SimpleNamespace(cwd=tmp_path)) + parent_task.prompt = parent_task.delegate_prompt + rows = _codex_parent_delegation_records( + initial_task, parent_task, "gpt-6-sol", "child-random-marker" + ) + + turn = completed_turn(CODEX, rows, parent_task) + + assert turn is not None + assert (turn.turn_id, turn.answer) == ("delegate", parent_task.value) + + +def test_cuj_evidence_codex_child_notification_or_echo_is_not_parent_completion(tmp_path): + initial_task = FileTask(SimpleNamespace(cwd=tmp_path)) + parent_task = FileTask(SimpleNamespace(cwd=tmp_path)) + parent_task.prompt = parent_task.delegate_prompt + rows = _codex_parent_delegation_records( + initial_task, parent_task, "gpt-6-sol", "child-random-marker" + ) + rows.pop() + + assert completed_turn(CODEX, rows, parent_task) is None + + +def test_cuj_evidence_codex_rejects_an_unrelated_later_parent_turn(): + task = SimpleNamespace(prompt="target", value="answer") + rows = records(CODEX, task, "model")[:-1] + rows.extend( + [ + { + "type": "event_msg", + "payload": {"type": "task_started", "turn_id": "unrelated"}, + }, + { + "type": "turn_context", + "payload": {"turn_id": "unrelated", "model": "model"}, + }, + { + "type": "response_item", + "payload": { + "type": "message", + "role": "user", + "content": [{"type": "input_text", "text": "different prompt"}], + }, + }, + { + "type": "event_msg", + "payload": { + "type": "task_complete", + "turn_id": "unrelated", + "last_agent_message": "answer", + }, + }, + ] + ) + + assert completed_turn(CODEX, rows, task) is None + + +def test_cuj_evidence_codex_keeps_completed_turn_before_later_turn(): + task = SimpleNamespace(prompt="target", value="answer") + rows = records(CODEX, task, "model") + rows.extend( + [ + { + "type": "event_msg", + "payload": {"type": "task_started", "turn_id": "later"}, + }, + { + "type": "turn_context", + "payload": {"turn_id": "later", "model": "model"}, + }, + { + "type": "event_msg", + "payload": {"type": "user_message", "message": "different prompt"}, + }, + { + "type": "event_msg", + "payload": { + "type": "task_complete", + "turn_id": "later", + "last_agent_message": "different answer", + }, + }, + ] + ) + + turn = completed_turn(CODEX, rows, task) + + assert turn is not None + assert (turn.turn_id, turn.answer) == ("turn", task.value) + + +def test_cuj_evidence_codex_pending_prompt_representations_are_not_duplicate(tmp_path): + task = FileTask(SimpleNamespace(cwd=tmp_path)) + rows = records(CODEX, task, "gpt-6-sol")[:-2] + rows.extend( + [ + { + "type": "event_msg", + "payload": {"type": "user_message", "message": task.prompt}, + }, + { + "type": "response_item", + "payload": { + "type": "message", + "role": "user", + "content": [{"type": "input_text", "text": task.prompt}], + }, + }, + ] + ) + + assert completed_turn(CODEX, rows, task) is None + + +def test_cuj_evidence_codex_rejects_repeated_prompt_representation(tmp_path): + task = FileTask(SimpleNamespace(cwd=tmp_path)) + rows = records(CODEX, task, "gpt-6-sol")[:-1] + rows.append(copy.deepcopy(rows[-1])) + + with pytest.raises(AssertionError, match="Prompt was submitted more than once"): + completed_turn(CODEX, rows, task) + + +def _inference_request(agent, sequence, payload): + return SimpleNamespace( + method="POST", + path=INFERENCE_PATHS[agent], + sequence=sequence, + payload=payload, + ) + + +def test_cuj_task_inference_selection_excludes_claude_title_and_helper_requests(): + prompt = "Read input-file.txt using a tool. Reply with only its contents." + title = _inference_request( + CLAUDE, + 1, + { + "model": "system.ai.claude-sonnet-4-6", + "messages": [{"role": "user", "content": f"Name this session: {prompt}"}], + "tools": [], + }, + ) + helper = _inference_request( + CLAUDE, + 2, + { + "model": "system.ai.claude-sonnet-4-6", + "messages": [{"role": "user", "content": prompt}], + "tools": [], + }, + ) + task = _inference_request( + CLAUDE, + 3, + { + "model": "system.ai.claude-haiku-4-5", + "messages": [{"role": "user", "content": prompt}], + "tools": [{"name": "Read"}], + }, + ) + + selected = _task_inference_request( + [title, helper, task], + CLAUDE, + prompt, + native_turn=SimpleNamespace(answer="file-value"), + ) + + assert selected is task + + +def test_cuj_child_inference_selection_excludes_codex_parent_continuation(): + parent_prompt = "Delegate this task to one subagent." + child_prompt = "Read input-file.txt using a tool and return its contents." + parent_continuation = _inference_request( + CODEX, + 4, + { + "model": "system.ai.gpt-6-sol", + "input": [ + {"role": "user", "content": [{"type": "input_text", "text": parent_prompt}]}, + { + "role": "assistant", + "content": [ + { + "type": "function_call", + "name": "spawn_agent", + "arguments": child_prompt, + } + ], + }, + {"role": "tool", "content": "child result for input-file.txt"}, + ], + "tools": [{"type": "function", "name": "spawn_agent"}], + }, + ) + child = _inference_request( + CODEX, + 5, + { + "model": "system.ai.gpt-6-luna", + "input": [{"role": "user", "content": [{"type": "input_text", "text": child_prompt}]}], + "tools": [{"type": "function", "name": "read_file"}], + }, + ) + + selected = _task_inference_request( + [parent_continuation, child], + CODEX, + child_prompt, + native_turn=SimpleNamespace(answer="child-result"), + native_prompts=(child_prompt,), + ) + + assert selected is child + + def test_cuj_evidence_ignores_existing_session(tmp_path): task = FileTask(SimpleNamespace(cwd=tmp_path)) first = SessionEvidence(tmp_path, CLAUDE) diff --git a/tests/test_smart_routing_config.py b/tests/test_smart_routing_config.py index 508fe5d1a..52987917b 100644 --- a/tests/test_smart_routing_config.py +++ b/tests/test_smart_routing_config.py @@ -10,7 +10,7 @@ ENABLE_SUBAGENT_ROUTING_ENV_VAR, SMART_ROUTER_CONFIG_VERSION_ENV_VAR, ) -from ucode.smart_routing import config +from ucode.smart_routing import config, orchestrator, v2 _EXPECTED_PRESETS = { config.FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0: { @@ -33,9 +33,29 @@ ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", }, + config.SUBAGENT_ORCH_V1: { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + }, } +@pytest.mark.parametrize("selector", ["unsupported_version", " unsupported_version "]) +def test_unknown_selector_is_a_no_op(selector): + applied = { + SMART_ROUTER_CONFIG_VERSION_ENV_VAR: selector, + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + "INHERITED_SETTING": "preserved", + } + original = applied.copy() + + assert config.apply_config(applied) == {} + assert applied == original + + @pytest.mark.parametrize("ENABLE_SMART_ROUTING_V2", [None, "0", "1"]) @pytest.mark.parametrize("ENABLE_SMART_ROUTING_SUBAGENT_ONLY", [None, "0", "1"]) @pytest.mark.parametrize("ENABLE_SMART_ROUTER_ORCHESTRATOR", [None, "0", "1"]) @@ -47,6 +67,7 @@ config.SUBAGENT_ONLY_V0, config.SUBAGENT_ONLY_V1, config.SUBAGENT_ORCH_V0, + config.SUBAGENT_ORCH_V1, ], ) def test_smart_routing_config_cartesian_grid( @@ -83,3 +104,151 @@ def test_smart_routing_config_cartesian_grid( config.apply_config(applied) assert applied == expected + + +@pytest.mark.parametrize( + ("legacy_env", "version", "expected_routing"), + [ + pytest.param( + { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + "SMART_ROUTER_NAME": "m2-r315-quality-20260929", + }, + config.SUBAGENT_ONLY_V1, + (True, False, False), + id="subagent-only-v1", + ), + pytest.param( + { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + "SMART_ROUTER_NAME": "m2-r315-quality-20260929", + }, + config.SUBAGENT_ORCH_V1, + (True, False, True), + id="subagent-orch-v1", + ), + pytest.param( + {ENABLE_SMART_ROUTING_ENV_VAR: "1"}, + config.FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0, + (True, True, False), + id="customer-first-prompt-and-subagent", + ), + ], +) +def test_legacy_preset_equivalence(legacy_env, version, expected_routing): + preset_env = {SMART_ROUTER_CONFIG_VERSION_ENV_VAR: version} + if "SMART_ROUTER_NAME" in legacy_env: + preset_env["SMART_ROUTER_NAME"] = legacy_env["SMART_ROUTER_NAME"] + original_legacy = legacy_env.copy() + original_preset = preset_env.copy() + expected_flags = _EXPECTED_PRESETS[version] + + assert {key: legacy_env.get(key, "0") for key in expected_flags} == expected_flags + + resolved = config.resolve_environment(preset_env) + applied = preset_env.copy() + config.apply_config(applied) + + for environment in (resolved, applied): + assert {key: environment[key] for key in expected_flags} == expected_flags + assert environment.get("SMART_ROUTER_NAME") == legacy_env.get("SMART_ROUTER_NAME") + assert SMART_ROUTER_CONFIG_VERSION_ENV_VAR not in environment + assert applied == resolved + + for environment in (legacy_env, preset_env, resolved, applied): + assert ( + v2.smart_routing_enabled(environment), + v2.first_prompt_routing_enabled(environment), + orchestrator.feature_enabled(environment), + ) == expected_routing + + assert legacy_env == original_legacy + assert preset_env == original_preset + + +@pytest.mark.parametrize("selector", [None, "", " "], ids=["missing", "empty", "whitespace"]) +def test_missing_selector_is_a_no_op(selector): + applied = { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + "INHERITED_SETTING": "preserved", + } + if selector is not None: + applied[SMART_ROUTER_CONFIG_VERSION_ENV_VAR] = selector + original = applied.copy() + + assert config.apply_config(applied) == {} + assert applied == original + + +@pytest.mark.parametrize("selector", ["unsupported_version", " unsupported_version "]) +def test_resolve_unknown_selector_omits_selector_without_mutating_input(selector): + source = { + SMART_ROUTER_CONFIG_VERSION_ENV_VAR: selector, + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + "INHERITED_SETTING": "preserved", + } + + assert config.resolve_environment(source) == { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + "INHERITED_SETTING": "preserved", + } + assert source == { + SMART_ROUTER_CONFIG_VERSION_ENV_VAR: selector, + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + "INHERITED_SETTING": "preserved", + } + + +@pytest.mark.parametrize(("version", "preset"), _EXPECTED_PRESETS.items()) +def test_resolve_known_selector_applies_preset_without_mutating_input(version, preset): + source = { + SMART_ROUTER_CONFIG_VERSION_ENV_VAR: version, + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + "INHERITED_SETTING": "preserved", + } + original = source.copy() + expected = source.copy() + expected.pop(SMART_ROUTER_CONFIG_VERSION_ENV_VAR) + expected.update(preset) + + assert config.resolve_environment(source) == expected + assert source == original + + +@pytest.mark.parametrize("version", _EXPECTED_PRESETS) +def test_apply_config_valid_selector_applies_and_restores(version): + original = { + SMART_ROUTER_CONFIG_VERSION_ENV_VAR: version, + ENABLE_SMART_ROUTING_ENV_VAR: "0", + "INHERITED_SETTING": "preserved", + } + expected = original.copy() + expected.pop(SMART_ROUTER_CONFIG_VERSION_ENV_VAR) + expected.update(_EXPECTED_PRESETS[version]) + applied = original.copy() + + previous = config.apply_config(applied) + + assert applied == expected + assert previous == { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: None, + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: None, + SMART_ROUTER_CONFIG_VERSION_ENV_VAR: version, + } + + v2.restore_smart_routing_env(previous, applied) + assert applied == original From 853c7ce25511e999d8da12c2bdcdd63ee40e3285 Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Fri, 9 Oct 2026 01:40:37 +0000 Subject: [PATCH 18/30] Trim routing config and CUJ regression coverage --- README.md | 19 +- tests/README.md | 17 +- tests/e2e_cuj/README.md | 10 +- tests/e2e_cuj/helpers/evidence.py | 137 ++++------ tests/e2e_cuj/test_cuj4_smart_routing.py | 135 +++------- tests/integration/README.md | 20 +- tests/test_cli.py | 22 +- tests/test_cuj_evidence.py | 304 +++++------------------ tests/test_smart_routing_config.py | 160 ++---------- 9 files changed, 194 insertions(+), 630 deletions(-) diff --git a/README.md b/README.md index bf5b5c2d2..1efe1cd7d 100644 --- a/README.md +++ b/README.md @@ -269,24 +269,7 @@ so first-prompt routing remains off; orchestration is also off. `subagent_orch_v1` enables all three legacy flags. Like `subagent_only_v1`, it routes subagents rather than the first prompt, and it additionally enables orchestration. -The following pairs select equivalent routing settings. Set the variables on the same command -line (or export them); separate unexported assignments joined by `&&` do not reliably reach UG. -Use the same `SMART_ROUTER_NAME` on both sides to select the same router. - -```bash -ENABLE_SMART_ROUTING_V2=1 ENABLE_SMART_ROUTING_SUBAGENT_ONLY=1 SMART_ROUTER_NAME=m2-r315-quality-20260929 uv run ug claude -SMART_ROUTER_CONFIG_VERSION=subagent_only_v1 SMART_ROUTER_NAME=m2-r315-quality-20260929 uv run ug claude - -ENABLE_SMART_ROUTING_V2=1 ENABLE_SMART_ROUTING_SUBAGENT_ONLY=1 ENABLE_SMART_ROUTER_ORCHESTRATOR=1 SMART_ROUTER_NAME=m2-r315-quality-20260929 uv run ug claude -SMART_ROUTER_CONFIG_VERSION=subagent_orch_v1 SMART_ROUTER_NAME=m2-r315-quality-20260929 uv run ug claude - -ENABLE_SMART_ROUTING_V2=1 uv run ug claude -SMART_ROUTER_CONFIG_VERSION=first_prompt_and_subagent_no_orch_v0 uv run ug claude -``` - -These pairs assume other routing flags are unset or `"0"`. The version selector overrides -inherited routing flags; legacy assignments leave unspecified flags inherited. -The orchestration flag is `ENABLE_SMART_ROUTER_ORCHESTRATOR`, not `ENABLE_SMART_ROUTING_ORCH`. +`SMART_ROUTER_NAME` still selects the router independently of the preset. Smart-routed Claude and Codex sessions install `smart-router`. The `subagent_orch_v0` and `subagent_orch_v1` versions also install and activate the bundled `smart-router-orchestrator` skill. diff --git a/tests/README.md b/tests/README.md index 741d51d2b..19062c943 100644 --- a/tests/README.md +++ b/tests/README.md @@ -25,10 +25,8 @@ the agent-specific prompt-submission evidence. Native evidence-reader regressions cover Claude's background-agent completion notifications; notifications alone cannot substitute for a final parent answer. These offline checks do not establish a live routing pass. -Offline CUJ regressions also cover Codex delegated-turn completion, including mirrored prompt -records and child notifications, while rejecting incomplete or unrelated turns. Smart-routing -request selection matches exact task prompts in tool-capable inference payloads, excluding -session-title requests and parent continuations when checking child inference. +Offline CUJ regressions cover Codex delegated-turn completion and exact task-request matching; +notifications alone, Claude title requests, and parent continuations do not qualify. The original smart-routing CUJ runs four fresh sessions: routed and explicit model for both Claude and Codex. One additional test has five `SMART_ROUTER_CONFIG_VERSION` cases. Each case exercises both agents and checks first-prompt routing, orchestrator context in @@ -153,14 +151,9 @@ routing flags take unset, `0`, and `1`, while the selector takes unset or one of supported presets, including the customer first-prompt-and-subagent mode. It uses the named version constants but independently hardcodes each preset's settings and asserts exact `resolve_environment` and `apply_config` settings, true-unset omission, unrelated-key and -input preservation, and valid-selector consumption. Additional helper cases cover no-op behavior -for unsupported selectors, valid-selector application/restoration, and unset/empty selectors. -CLI regressions in `test_cli.py` verify that unknown selectors do not block token-only -authentication, revert, or Claude/Codex/default launch dispatch, and preserve the environment. -Import-time schema validation, -routing getters, native subcommands, hooks, -session files, or managed launches are not established by this grid. These are component -checks; they do not establish live agent, hook, or gateway behavior. +input preservation, valid-selector consumption, and environment restoration. +Absent/unknown selectors are no-ops; CLI checks cover auth, revert, and launch dispatch. +These component checks do not establish live agent, hook, or gateway behavior. Three additional component cases compare the legacy flags with `subagent_only_v1`, `subagent_orch_v1`, and `first_prompt_and_subagent_no_orch_v0`, including first-prompt, subagent-routing activation, orchestration, and preservation of the chosen router name. diff --git a/tests/e2e_cuj/README.md b/tests/e2e_cuj/README.md index 88e349957..7908ee0be 100644 --- a/tests/e2e_cuj/README.md +++ b/tests/e2e_cuj/README.md @@ -76,16 +76,14 @@ the expected presence/absence of a prompt-correlated router request, successful the routed or configured default model, completed native file-task evidence, and no child session. Orchestrator presence means its activation context reached the real gateway inference input, not that an assistant echoed it or a skill merely existed on disk. -Task inference is selected by its exact user prompt and tool-capable payload, not the first -POST: session-title requests are excluded. Child requests use the routed prompt from the native -child transcript, so parent continuations cannot stand in for child inference. +Task inference must contain the exact task/routed prompt and tools, excluding Claude title +requests and parent continuations from child-inference checks. Each preset session then explicitly requests one child for a separate hidden-value file task. Assertions require a native child transcript containing the value, the completed parent answer, a correlated spawn-routing decision, and successful child inference on the router's selected model. This tests requested delegation, not automatic orchestrator delegation. -Codex completion evidence tolerates mirrored prompt records and child notifications but still -requires the matching native completed turn and final parent answer. Offline regressions cover -these transcript and request-selection cases; they do not establish a live CUJ pass. +Codex requires a matching completed turn and final parent answer, not just child notifications. +Offline evidence regressions do not establish a live CUJ pass. Selector cases have separate TUI artifact names, and the session environment is restored afterward. Each preset also runs public `ug revert` in cleanup, including after a failed assertion, so interactive launches' OS-managed settings cannot contaminate the next preset's configuration. diff --git a/tests/e2e_cuj/helpers/evidence.py b/tests/e2e_cuj/helpers/evidence.py index 6db404ad7..00de1cd0a 100644 --- a/tests/e2e_cuj/helpers/evidence.py +++ b/tests/e2e_cuj/helpers/evidence.py @@ -154,118 +154,77 @@ def _completed_turn(records, task): meta = [row["payload"] for row in records if row.get("type") == "session_meta"] if len(meta) != 1 or isinstance(meta[0].get("source"), dict): return None - contexts, started, completed, aborted, current_turn = {}, set(), set(), set(), None - target_turn, prompt_sources, seen_prompt, final_answer = None, set(), False, None - - def notification_text(row): - payload = row.get("payload", {}) - if row.get("type") == "event_msg" and payload.get("type") == "user_message": - return payload.get("message", "") - if ( - row.get("type") == "response_item" - and payload.get("type") == "message" - and payload.get("role") == "user" - ): - return message_text(payload.get("content")) - return "" - - def is_notification(row): - text = notification_text(row) - normalized = text.lower() if isinstance(text, str) else "" - return ( - "" in normalized or "" in normalized - ) - - def row_turn(row): - payload = row.get("payload", {}) - if payload.get("turn_id"): - return payload["turn_id"] - metadata = row.get("internal_chat_message_metadata_passthrough") - return metadata.get("turn_id") if isinstance(metadata, dict) else None - - def handle_prompt(row, source): - nonlocal seen_prompt, target_turn - payload = row.get("payload", {}) - if row.get("type") == "response_item": - prompt = message_text(payload.get("content")) - else: - prompt = payload.get("message") - if prompt != task.prompt: - return False - if source in prompt_sources: - raise AssertionError("Prompt was submitted more than once") - prompt_sources.add(source) - if not seen_prompt: - seen_prompt = True - prompt_turn = row_turn(row) or current_turn - if target_turn is None and prompt_turn not in completed | aborted: - target_turn = prompt_turn - elif target_turn is not None and prompt_turn is not None: - assert prompt_turn == target_turn, "Prompt was submitted more than once" - return True - + contexts, active_turn, target_turn = {}, None, None + started_turn, aborted_turn, answer, prompt_sources = None, None, None, set() for row in records: payload = row.get("payload", {}) + metadata = row.get("internal_chat_message_metadata_passthrough") or {} + if "multi_agent.subagent_notification" in metadata.get( + "content_item_kinds", () + ) or "" in str( + payload.get("content") or payload.get("message") + ).replace("-", "_"): + continue + row_turn = payload.get("turn_id") if row.get("type") == "turn_context": - turn_id = payload.get("turn_id") - contexts.setdefault(turn_id, []).append(payload.get("model")) - current_turn = turn_id - if seen_prompt and target_turn is None and turn_id not in completed | aborted: - target_turn = turn_id - elif target_turn is not None and turn_id != target_turn: + contexts.setdefault(row_turn, []).append(payload.get("model")) + if target_turn and row_turn != target_turn: return None - if ( + active_turn = row_turn + continue + source = None + if row.get("type") == "event_msg" and payload.get("type") == "user_message": + source, prompt = "event", payload.get("message") + elif ( row.get("type") == "response_item" and payload.get("type") == "message" and payload.get("role") == "user" ): - if handle_prompt(row, "response_item"): - continue - if seen_prompt and not is_notification(row): + source, prompt = "response", message_text(payload.get("content")) + if source: + if prompt == task.prompt: + assert source not in prompt_sources, "Prompt was submitted more than once" + prompt_sources.add(source) + if target_turn and row_turn and row_turn != target_turn: + return None + target_turn = target_turn or row_turn or active_turn + elif prompt_sources: return None + continue if row.get("type") != "event_msg": continue event_type = payload.get("type") if event_type == "task_started": turn_id = payload.get("turn_id") - if not turn_id: - continue - started.add(turn_id) - current_turn = turn_id - if not seen_prompt: - continue - if target_turn is None and turn_id not in completed | aborted: - target_turn = turn_id - elif turn_id != target_turn: + if not turn_id or turn_id == aborted_turn: return None - elif event_type == "user_message": - if handle_prompt(row, "user_message"): - continue - if seen_prompt and not is_notification(row): + if target_turn and turn_id != target_turn: return None + active_turn = started_turn = turn_id + if prompt_sources and target_turn is None: + target_turn = turn_id elif event_type == "task_complete": turn_id = payload.get("turn_id") - completed.add(turn_id) - if seen_prompt and turn_id == target_turn: - final_answer = payload.get("last_agent_message") or "" + if target_turn: + if turn_id != target_turn: + return None + answer = payload.get("last_agent_message") or "" break + if turn_id == active_turn: + active_turn = started_turn = None elif event_type == "turn_aborted": - turn_id = payload.get("turn_id") - aborted.add(turn_id) - if turn_id == target_turn: + aborted_turn = payload.get("turn_id") + if aborted_turn == target_turn: return None - - if ( - not seen_prompt - or target_turn is None - or target_turn not in started - or final_answer is None - ): + if aborted_turn == active_turn: + active_turn = started_turn = None + if not prompt_sources or not target_turn or started_turn != target_turn or answer is None: return None - if task.value not in final_answer or not contexts.get(target_turn): + models = contexts.get(target_turn) + if task.value not in answer or not models: return None - assert all(contexts[target_turn]), "Missing native turn model metadata" - return CompletedTurn(meta[0]["id"], target_turn, contexts[target_turn], final_answer) + assert all(models), "Missing native turn model metadata" + return CompletedTurn(meta[0]["id"], target_turn, models, answer) def get_cuj_helper(agent): diff --git a/tests/e2e_cuj/test_cuj4_smart_routing.py b/tests/e2e_cuj/test_cuj4_smart_routing.py index ce86f6d20..618648fc1 100644 --- a/tests/e2e_cuj/test_cuj4_smart_routing.py +++ b/tests/e2e_cuj/test_cuj4_smart_routing.py @@ -27,7 +27,6 @@ SessionObservation, canonical_model, claude_file_task, - message_text, ) from .helpers.terminal import Terminal from .helpers.tui_request_recorder import RecordedRequest, RecordedResponse @@ -87,75 +86,40 @@ def _run_session(session, recorder, agent, task, launch_args): return evidence.observe(task), recorder.requests_after(checkpoint) -def _content_is_exact_prompt(content, prompt): - if isinstance(content, str): - return content == prompt - return isinstance(content, list) and any( - isinstance(part, dict) - and part.get("type") in {"text", "input_text"} - and part.get("text") == prompt - for part in content - ) - - -def _request_contains_exact_user_prompt(request, agent, prompt): - field = "messages" if agent == CLAUDE else "input" - entries = request.payload.get(field) - if isinstance(entries, str): - return entries == prompt - if not isinstance(entries, list): - return False - return any( - isinstance(entry, dict) - and entry.get("role") == "user" - and _content_is_exact_prompt(entry.get("content"), prompt) - for entry in entries - ) - - -def _is_tool_capable(request): - tools = request.payload.get("tools") - return isinstance(tools, list) and bool(tools) - - -def _task_inference_request(requests, agent, prompt, *, native_turn, after=0, native_prompts=()): - """Select a task request from payload evidence, never from its expected model.""" - assert native_turn is not None, "No completed native turn anchors the task request" - if native_prompts: - assert prompt in native_prompts, "Routed prompt is absent from the native child session" - matches = tuple( - request - for request in requests - if request.sequence > after - and request.method == "POST" - and request.path == INFERENCE_PATHS[agent] - and _request_contains_exact_user_prompt(request, agent, prompt) - and _is_tool_capable(request) - ) - assert matches, "No tool-capable inference request contained the exact task prompt" - return matches[0] - - -def _native_user_prompts(agent, records): - prompts = [] - for record in records: - if agent == CLAUDE and record.get("type") == "user": - prompt = message_text(record.get("message", {}).get("content")) - elif agent == CODEX and record.get("type") == "event_msg": - payload = record.get("payload", {}) - prompt = payload.get("message") if payload.get("type") == "user_message" else "" - elif agent == CODEX and record.get("type") == "response_item": - payload = record.get("payload", {}) - prompt = ( - message_text(payload.get("content")) - if payload.get("type") == "message" and payload.get("role") == "user" - else "" - ) - else: - prompt = "" - if isinstance(prompt, str) and prompt: - prompts.append(prompt) - return tuple(dict.fromkeys(prompts)) +def _task_inference_request(requests, agent, prompt, *, after=0): + """Match the task payload, not the expected model or orchestrator context.""" + for request in requests: + tools = request.payload.get("tools") + if ( + request.sequence <= after + or request.method != "POST" + or request.path != INFERENCE_PATHS[agent] + or not isinstance(tools, list) + or not tools + ): + continue + entries = request.payload.get("messages" if agent == CLAUDE else "input", []) + if isinstance(entries, str): + if entries == prompt: + return request + continue + if not isinstance(entries, list): + continue + for entry in entries: + if not isinstance(entry, dict) or entry.get("role") != "user": + continue + content = entry.get("content", []) + if isinstance(content, str): + if content == prompt: + return request + elif isinstance(content, list) and any( + isinstance(part, dict) + and part.get("type") in {"text", "input_text"} + and part.get("text") == prompt + for part in content + ): + return request + raise AssertionError("No tool-capable inference request contained the exact task prompt") def _assert_published_config_matches_expectations(published): @@ -212,12 +176,7 @@ def run_smart_routing_journeys(cuj) -> SmartRoutingSessionResults: if request.method == "POST" and request.path == ROUTING_PATH ) route_response = recorder.response_for(route_request) - inference_request = _task_inference_request( - requests, - agent, - task.prompt, - native_turn=observation.turn, - ) + inference_request = _task_inference_request(requests, agent, task.prompt) no_model_override[agent] = SessionCase( agent=agent, launch_args=(agent,), @@ -233,12 +192,7 @@ def run_smart_routing_journeys(cuj) -> SmartRoutingSessionResults: task = _file_task(session, agent) launch_args = (agent, "--model", overrides[agent]) observation, requests = _run_session(session, recorder, agent, task, launch_args) - inference_request = _task_inference_request( - requests, - agent, - task.prompt, - native_turn=observation.turn, - ) + inference_request = _task_inference_request(requests, agent, task.prompt) with_model_override[agent] = SessionCase( agent=agent, launch_args=launch_args, @@ -314,13 +268,7 @@ def test_smart_router_config_version( selections = response.payload["route_selection"] assert len(selections) == 1 expected_model = selections[0]["route_option"]["model"] - native_turn = evidence.completed(task) - inference = _task_inference_request( - requests, - agent, - task.prompt, - native_turn=native_turn, - ) + inference = _task_inference_request(requests, agent, task.prompt) assert recorder.response_for(inference).status_code == 200 assert canonical_model(inference.payload["model"]) == canonical_model( expected_model @@ -362,12 +310,6 @@ def test_smart_router_config_version( for records in children.values() for answer in assistant_answers(agent, records) ), agent - native_turn = evidence.completed(child_task) - child_prompts = tuple( - prompt - for records in children.values() - for prompt in _native_user_prompts(agent, records) - ) decisions = read_jsonl(decisions_path)[decision_count:] assert_subagent_routed( session, @@ -379,7 +321,6 @@ def test_smart_router_config_version( routes = [request for request in requests if request.path == ROUTING_PATH] assert len(routes) == 1, (agent, routes) route_prompt = routes[0].payload["task"]["prompt"] - assert route_prompt in child_prompts, (agent, route_prompt, child_prompts) assert child_task.filename in route_prompt response = recorder.response_for(routes[0]) assert response.status_code == 200 @@ -389,9 +330,7 @@ def test_smart_router_config_version( requests, agent, route_prompt, - native_turn=native_turn, after=routes[0].sequence, - native_prompts=child_prompts, ) assert recorder.response_for(inference).status_code == 200 assert canonical_model(inference.payload["model"]) == canonical_model( diff --git a/tests/integration/README.md b/tests/integration/README.md index 0d57262ea..035fd4ba6 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -18,10 +18,8 @@ This is not automatic orchestrator-delegation coverage. No workspace configurati Each preset cleans up interactive OS-managed settings with public `ug revert`, even on failure. Offline transcript tests check native Claude background-agent completion evidence; a completion notification alone is not treated as the parent's completed answer. -Offline CUJ cases additionally cover Codex delegated-turn completion and reject incomplete -or unrelated turns. Smart-routing requests are matched by exact task prompts and tool-capable -payloads rather than the first inference POST, excluding session-title and parent-continuation -traffic without weakening model, orchestrator-context, or paired-response assertions. +Offline CUJ cases also cover Codex delegated-turn completion and task-request matching; +see `../e2e_cuj/README.md` for the evidence requirements. The [catalog discovery journey](../e2e_cuj/README.md) uses the CUJ3 workspace to check agent-compatible pickers, schema exclusions, configured defaults, and real inference. @@ -325,16 +323,12 @@ and on/off coverage; preset-specific behavior there is not covered. The unit/component `../test_smart_routing_config.py` includes a 162-case Cartesian oracle over all three legacy routing flags (`None`, `0`, `1`) and six selector forms (`None` plus the five -supported presets, including the customer first-prompt-and-subagent mode). It independently hardcodes preset values and asserts exact +supported presets, including the customer first-prompt-and-subagent mode). It independently +hardcodes preset values and asserts exact `resolve_environment` and `apply_config` settings, true-unset omission, unrelated-key and -input preservation, and valid-selector consumption. Additional helper cases cover no-op behavior -for unsupported selectors, valid-selector application/restoration, and unset/empty selectors. -Component cases in `../test_cli.py` check token-only authentication, revert, and -Claude/Codex/default launch dispatch with unknown selectors and environment preservation. -Import-time schema validation, -routing getters, native subcommands, hooks, session -files, or managed launches are not established here. It does not claim live agent, hook, or -gateway coverage. +input preservation, valid-selector consumption, and environment restoration. +Absent/unknown selectors are no-ops; CLI checks cover auth, revert, and launch dispatch. +These component checks do not establish live agent, hook, or gateway behavior. Three component cases additionally compare the equivalent legacy flags and presets for `subagent_only_v1`, `subagent_orch_v1`, and `first_prompt_and_subagent_no_orch_v0`; they verify routing activation, first-prompt behavior, orchestration, and router-name preservation. diff --git a/tests/test_cli.py b/tests/test_cli.py index f1d1e7d54..b95af40d1 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -1875,7 +1875,11 @@ def _isolated_bearer(self): else: os.environ["DATABRICKS_BEARER"] = original - def test_prints_only_the_token_to_stdout(self): + @pytest.mark.parametrize("selector", [None, "unsupported_v99"]) + def test_prints_only_the_token_to_stdout(self, monkeypatch, selector): + if selector is not None: + monkeypatch.setenv("SMART_ROUTER_CONFIG_VERSION", selector) + previous = dict(os.environ) with ( patch("ucode.cli.load_state", return_value={"workspace": "https://ws"}), patch("ucode.cli.get_databricks_token", return_value="tok-123") as fetch, @@ -1886,6 +1890,7 @@ def test_prints_only_the_token_to_stdout(self): # or the consuming agent will treat the noise as part of the token. assert result.stdout == "tok-123\n" fetch.assert_called_once_with("https://ws", None, force_refresh=False) + assert dict(os.environ) == previous def test_host_and_profile_override_state(self): with ( @@ -1898,21 +1903,6 @@ def test_host_and_profile_override_state(self): assert result.exit_code == 0 fetch.assert_called_once_with("https://override", "prod", force_refresh=False) - def test_invalid_routing_config_does_not_block_token(self, monkeypatch): - monkeypatch.setenv("SMART_ROUTER_CONFIG_VERSION", "unsupported_v99") - monkeypatch.setenv("ENABLE_SMART_ROUTING_V2", "1") - previous = dict(os.environ) - with ( - patch("ucode.cli.load_state", return_value={"workspace": "https://ws"}), - patch("ucode.cli.get_databricks_token", return_value="tok-123") as fetch, - ): - result = runner.invoke(app, ["auth-token"]) - - assert result.exit_code == 0, result.output - assert result.stdout == "tok-123\n" - fetch.assert_called_once_with("https://ws", None, force_refresh=False) - assert dict(os.environ) == previous - def test_force_refresh_is_forwarded(self): with ( patch("ucode.cli.load_state", return_value={"workspace": "https://ws"}), diff --git a/tests/test_cuj_evidence.py b/tests/test_cuj_evidence.py index a58644fbf..a5b362b6f 100644 --- a/tests/test_cuj_evidence.py +++ b/tests/test_cuj_evidence.py @@ -228,273 +228,93 @@ def test_cuj_evidence_codex_rejects_wrong_turn(tmp_path): assert completed_turn(CODEX, wrong_turn, task) is None -def _codex_parent_delegation_records(initial_task, parent_task, model, child_marker): - rows = records(CODEX, initial_task, model) - rows.extend( - [ - { - "type": "event_msg", - "payload": {"type": "user_message", "message": parent_task.prompt}, - }, - {"type": "event_msg", "payload": {"type": "task_started", "turn_id": "delegate"}}, - { - "type": "turn_context", - "payload": {"turn_id": "delegate", "model": model}, - }, - { - "type": "response_item", - "payload": { - "type": "message", - "role": "user", - "content": [ - { - "type": "input_text", - "text": ( - "\n" - f'{{"agent_path":"child","status":{{"completed":"{child_marker}"}}}}\n' - "" - ), - } - ], - }, - "internal_chat_message_metadata_passthrough": { - "turn_id": "delegate", - "content_item_kinds": ["multi_agent.subagent_notification"], - }, - }, - { - "type": "event_msg", - "payload": { - "type": "task_complete", - "turn_id": "delegate", - "last_agent_message": parent_task.value, - }, - }, - ] - ) - return rows - - -def test_cuj_evidence_codex_parent_turn_survives_child_notification(tmp_path): - initial_task = FileTask(SimpleNamespace(cwd=tmp_path)) - parent_task = FileTask(SimpleNamespace(cwd=tmp_path)) - parent_task.prompt = parent_task.delegate_prompt - rows = _codex_parent_delegation_records( - initial_task, parent_task, "gpt-6-sol", "child-random-marker" - ) - - turn = completed_turn(CODEX, rows, parent_task) - - assert turn is not None - assert (turn.turn_id, turn.answer) == ("delegate", parent_task.value) - - -def test_cuj_evidence_codex_child_notification_or_echo_is_not_parent_completion(tmp_path): - initial_task = FileTask(SimpleNamespace(cwd=tmp_path)) - parent_task = FileTask(SimpleNamespace(cwd=tmp_path)) - parent_task.prompt = parent_task.delegate_prompt - rows = _codex_parent_delegation_records( - initial_task, parent_task, "gpt-6-sol", "child-random-marker" - ) - rows.pop() - - assert completed_turn(CODEX, rows, parent_task) is None +def _tagged_codex_notification(row): + notification = copy.deepcopy(row) + notification["payload"]["content"] = [{"type": "input_text", "text": ""}] + notification["internal_chat_message_metadata_passthrough"] = { + "content_item_kinds": ["multi_agent.subagent_notification"] + } + return notification -def test_cuj_evidence_codex_rejects_an_unrelated_later_parent_turn(): - task = SimpleNamespace(prompt="target", value="answer") - rows = records(CODEX, task, "model")[:-1] +def _codex_parent_delegation_records(task, model): + rows = records(CODEX, SimpleNamespace(prompt="initial", value="initial"), model) + target = records(CODEX, task, model)[1:] + for row in target: + if row["type"] != "response_item": + row["payload"]["turn_id"] = "delegate" rows.extend( [ - { - "type": "event_msg", - "payload": {"type": "task_started", "turn_id": "unrelated"}, - }, - { - "type": "turn_context", - "payload": {"turn_id": "unrelated", "model": "model"}, - }, - { - "type": "response_item", - "payload": { - "type": "message", - "role": "user", - "content": [{"type": "input_text", "text": "different prompt"}], - }, - }, - { - "type": "event_msg", - "payload": { - "type": "task_complete", - "turn_id": "unrelated", - "last_agent_message": "answer", - }, - }, + {"type": "event_msg", "payload": {"type": "user_message", "message": task.prompt}}, + *target[:3], + _tagged_codex_notification(target[2]), + target[3], ] ) + return rows - assert completed_turn(CODEX, rows, task) is None - - -def test_cuj_evidence_codex_keeps_completed_turn_before_later_turn(): - task = SimpleNamespace(prompt="target", value="answer") - rows = records(CODEX, task, "model") - rows.extend( - [ - { - "type": "event_msg", - "payload": {"type": "task_started", "turn_id": "later"}, - }, - { - "type": "turn_context", - "payload": {"turn_id": "later", "model": "model"}, - }, - { - "type": "event_msg", - "payload": {"type": "user_message", "message": "different prompt"}, - }, - { - "type": "event_msg", - "payload": { - "type": "task_complete", - "turn_id": "later", - "last_agent_message": "different answer", - }, - }, - ] - ) +def test_cuj_evidence_codex_delegated_parent_turn(tmp_path): + task = FileTask(SimpleNamespace(cwd=tmp_path)) + task.prompt = task.delegate_prompt + rows = _codex_parent_delegation_records(task, "gpt-6-sol") + later = records(CODEX, SimpleNamespace(prompt="later", value="later"), "gpt-6-sol")[1:] + for row in later: + if row["type"] != "response_item": + row["payload"]["turn_id"] = "later" + rows.extend(later) turn = completed_turn(CODEX, rows, task) - assert turn is not None - assert (turn.turn_id, turn.answer) == ("turn", task.value) - - -def test_cuj_evidence_codex_pending_prompt_representations_are_not_duplicate(tmp_path): - task = FileTask(SimpleNamespace(cwd=tmp_path)) - rows = records(CODEX, task, "gpt-6-sol")[:-2] - rows.extend( - [ - { - "type": "event_msg", - "payload": {"type": "user_message", "message": task.prompt}, - }, - { - "type": "response_item", - "payload": { - "type": "message", - "role": "user", - "content": [{"type": "input_text", "text": task.prompt}], - }, - }, - ] - ) - - assert completed_turn(CODEX, rows, task) is None + assert (turn.turn_id, turn.answer) == ("delegate", task.value) -def test_cuj_evidence_codex_rejects_repeated_prompt_representation(tmp_path): +@pytest.mark.parametrize("case", ["aborted", "later", "notification", "duplicate"]) +def test_cuj_evidence_codex_rejects_follow_up_false_positives(tmp_path, case): task = FileTask(SimpleNamespace(cwd=tmp_path)) - rows = records(CODEX, task, "gpt-6-sol")[:-1] - rows.append(copy.deepcopy(rows[-1])) - - with pytest.raises(AssertionError, match="Prompt was submitted more than once"): - completed_turn(CODEX, rows, task) + rows = records(CODEX, task, "gpt-6-sol") + if case == "aborted": + rows[-1]["payload"]["type"] = "turn_aborted" + elif case == "later": + rows.pop() + rows.append({"type": "event_msg", "payload": {"type": "task_started", "turn_id": "later"}}) + elif case == "notification": + rows[3] = _tagged_codex_notification(rows[3]) + else: + rows.insert(-1, copy.deepcopy(rows[3])) + if case == "duplicate": + with pytest.raises(AssertionError, match="Prompt was submitted more than once"): + completed_turn(CODEX, rows, task) + else: + assert completed_turn(CODEX, rows, task) is None -def _inference_request(agent, sequence, payload): +def _inference_request(agent, sequence, prompt, tools): + field = "messages" if agent == CLAUDE else "input" + content = prompt if agent == CLAUDE else [{"type": "input_text", "text": prompt}] return SimpleNamespace( method="POST", path=INFERENCE_PATHS[agent], sequence=sequence, - payload=payload, + payload={field: [{"role": "user", "content": content}], "tools": tools}, ) -def test_cuj_task_inference_selection_excludes_claude_title_and_helper_requests(): +def test_cuj_task_inference_selection_uses_exact_tool_prompt(): prompt = "Read input-file.txt using a tool. Reply with only its contents." - title = _inference_request( - CLAUDE, - 1, - { - "model": "system.ai.claude-sonnet-4-6", - "messages": [{"role": "user", "content": f"Name this session: {prompt}"}], - "tools": [], - }, - ) - helper = _inference_request( - CLAUDE, - 2, - { - "model": "system.ai.claude-sonnet-4-6", - "messages": [{"role": "user", "content": prompt}], - "tools": [], - }, - ) - task = _inference_request( - CLAUDE, - 3, - { - "model": "system.ai.claude-haiku-4-5", - "messages": [{"role": "user", "content": prompt}], - "tools": [{"name": "Read"}], - }, - ) - - selected = _task_inference_request( - [title, helper, task], - CLAUDE, - prompt, - native_turn=SimpleNamespace(answer="file-value"), - ) - - assert selected is task - - -def test_cuj_child_inference_selection_excludes_codex_parent_continuation(): - parent_prompt = "Delegate this task to one subagent." + claude = [ + _inference_request(CLAUDE, 1, f"Name this session: {prompt}", []), + _inference_request(CLAUDE, 2, prompt, []), + _inference_request(CLAUDE, 3, prompt, [{"name": "Read"}]), + ] + assert _task_inference_request(claude, CLAUDE, prompt) is claude[2] child_prompt = "Read input-file.txt using a tool and return its contents." - parent_continuation = _inference_request( - CODEX, - 4, - { - "model": "system.ai.gpt-6-sol", - "input": [ - {"role": "user", "content": [{"type": "input_text", "text": parent_prompt}]}, - { - "role": "assistant", - "content": [ - { - "type": "function_call", - "name": "spawn_agent", - "arguments": child_prompt, - } - ], - }, - {"role": "tool", "content": "child result for input-file.txt"}, - ], - "tools": [{"type": "function", "name": "spawn_agent"}], - }, - ) - child = _inference_request( - CODEX, - 5, - { - "model": "system.ai.gpt-6-luna", - "input": [{"role": "user", "content": [{"type": "input_text", "text": child_prompt}]}], - "tools": [{"type": "function", "name": "read_file"}], - }, + parent = _inference_request( + CODEX, 4, "Delegate this task to one subagent.", [{"name": "spawn_agent"}] ) - - selected = _task_inference_request( - [parent_continuation, child], - CODEX, - child_prompt, - native_turn=SimpleNamespace(answer="child-result"), - native_prompts=(child_prompt,), - ) - - assert selected is child + child = _inference_request(CODEX, 5, child_prompt, [{"name": "read_file"}]) + assert _task_inference_request([parent, child], CODEX, child_prompt, after=3) is child + with pytest.raises(AssertionError, match="No tool-capable"): + _task_inference_request([parent, child], CODEX, child_prompt, after=5) def test_cuj_evidence_ignores_existing_session(tmp_path): diff --git a/tests/test_smart_routing_config.py b/tests/test_smart_routing_config.py index 52987917b..c797902a0 100644 --- a/tests/test_smart_routing_config.py +++ b/tests/test_smart_routing_config.py @@ -41,17 +41,23 @@ } -@pytest.mark.parametrize("selector", ["unsupported_version", " unsupported_version "]) -def test_unknown_selector_is_a_no_op(selector): +@pytest.mark.parametrize( + "selector", [None, "", " ", "unsupported_version", " unsupported_version "] +) +def test_absent_or_unknown_selector_is_a_no_op(selector): applied = { - SMART_ROUTER_CONFIG_VERSION_ENV_VAR: selector, ENABLE_SMART_ROUTING_ENV_VAR: "1", ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", "INHERITED_SETTING": "preserved", } + expected = applied.copy() + if selector is not None: + applied[SMART_ROUTER_CONFIG_VERSION_ENV_VAR] = selector original = applied.copy() + assert config.resolve_environment(applied) == expected + assert applied == original assert config.apply_config(applied) == {} assert applied == original @@ -101,154 +107,36 @@ def test_smart_routing_config_cartesian_grid( assert source == original applied = source.copy() - config.apply_config(applied) + previous = config.apply_config(applied) assert applied == expected + v2.restore_smart_routing_env(previous, applied) + assert applied == original @pytest.mark.parametrize( - ("legacy_env", "version", "expected_routing"), + ("version", "expected_routing"), [ - pytest.param( - { - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", - "SMART_ROUTER_NAME": "m2-r315-quality-20260929", - }, - config.SUBAGENT_ONLY_V1, - (True, False, False), - id="subagent-only-v1", - ), - pytest.param( - { - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", - "SMART_ROUTER_NAME": "m2-r315-quality-20260929", - }, - config.SUBAGENT_ORCH_V1, - (True, False, True), - id="subagent-orch-v1", - ), - pytest.param( - {ENABLE_SMART_ROUTING_ENV_VAR: "1"}, - config.FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0, - (True, True, False), - id="customer-first-prompt-and-subagent", - ), + (config.SUBAGENT_ONLY_V1, (True, False, False)), + (config.SUBAGENT_ORCH_V1, (True, False, True)), + (config.FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0, (True, True, False)), ], ) -def test_legacy_preset_equivalence(legacy_env, version, expected_routing): - preset_env = {SMART_ROUTER_CONFIG_VERSION_ENV_VAR: version} - if "SMART_ROUTER_NAME" in legacy_env: - preset_env["SMART_ROUTER_NAME"] = legacy_env["SMART_ROUTER_NAME"] - original_legacy = legacy_env.copy() - original_preset = preset_env.copy() - expected_flags = _EXPECTED_PRESETS[version] - - assert {key: legacy_env.get(key, "0") for key in expected_flags} == expected_flags - +def test_legacy_preset_equivalence(version, expected_routing): + legacy_env = _EXPECTED_PRESETS[version].copy() + legacy_env["SMART_ROUTER_NAME"] = "m2-r315-quality-20260929" + preset_env = { + SMART_ROUTER_CONFIG_VERSION_ENV_VAR: version, + "SMART_ROUTER_NAME": legacy_env["SMART_ROUTER_NAME"], + } resolved = config.resolve_environment(preset_env) applied = preset_env.copy() config.apply_config(applied) - - for environment in (resolved, applied): - assert {key: environment[key] for key in expected_flags} == expected_flags - assert environment.get("SMART_ROUTER_NAME") == legacy_env.get("SMART_ROUTER_NAME") - assert SMART_ROUTER_CONFIG_VERSION_ENV_VAR not in environment assert applied == resolved - for environment in (legacy_env, preset_env, resolved, applied): + assert environment["SMART_ROUTER_NAME"] == legacy_env["SMART_ROUTER_NAME"] assert ( v2.smart_routing_enabled(environment), v2.first_prompt_routing_enabled(environment), orchestrator.feature_enabled(environment), ) == expected_routing - - assert legacy_env == original_legacy - assert preset_env == original_preset - - -@pytest.mark.parametrize("selector", [None, "", " "], ids=["missing", "empty", "whitespace"]) -def test_missing_selector_is_a_no_op(selector): - applied = { - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", - "INHERITED_SETTING": "preserved", - } - if selector is not None: - applied[SMART_ROUTER_CONFIG_VERSION_ENV_VAR] = selector - original = applied.copy() - - assert config.apply_config(applied) == {} - assert applied == original - - -@pytest.mark.parametrize("selector", ["unsupported_version", " unsupported_version "]) -def test_resolve_unknown_selector_omits_selector_without_mutating_input(selector): - source = { - SMART_ROUTER_CONFIG_VERSION_ENV_VAR: selector, - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", - "INHERITED_SETTING": "preserved", - } - - assert config.resolve_environment(source) == { - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", - "INHERITED_SETTING": "preserved", - } - assert source == { - SMART_ROUTER_CONFIG_VERSION_ENV_VAR: selector, - ENABLE_SMART_ROUTING_ENV_VAR: "1", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", - "INHERITED_SETTING": "preserved", - } - - -@pytest.mark.parametrize(("version", "preset"), _EXPECTED_PRESETS.items()) -def test_resolve_known_selector_applies_preset_without_mutating_input(version, preset): - source = { - SMART_ROUTER_CONFIG_VERSION_ENV_VAR: version, - ENABLE_SMART_ROUTING_ENV_VAR: "0", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", - "INHERITED_SETTING": "preserved", - } - original = source.copy() - expected = source.copy() - expected.pop(SMART_ROUTER_CONFIG_VERSION_ENV_VAR) - expected.update(preset) - - assert config.resolve_environment(source) == expected - assert source == original - - -@pytest.mark.parametrize("version", _EXPECTED_PRESETS) -def test_apply_config_valid_selector_applies_and_restores(version): - original = { - SMART_ROUTER_CONFIG_VERSION_ENV_VAR: version, - ENABLE_SMART_ROUTING_ENV_VAR: "0", - "INHERITED_SETTING": "preserved", - } - expected = original.copy() - expected.pop(SMART_ROUTER_CONFIG_VERSION_ENV_VAR) - expected.update(_EXPECTED_PRESETS[version]) - applied = original.copy() - - previous = config.apply_config(applied) - - assert applied == expected - assert previous == { - ENABLE_SMART_ROUTING_ENV_VAR: "0", - ENABLE_SUBAGENT_ROUTING_ENV_VAR: None, - ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: None, - SMART_ROUTER_CONFIG_VERSION_ENV_VAR: version, - } - - v2.restore_smart_routing_env(previous, applied) - assert applied == original From aa926c408f749df5aabde1f96d1b789b046e4765 Mon Sep 17 00:00:00 2001 From: Lilly Luo Date: Fri, 9 Oct 2026 01:51:20 +0000 Subject: [PATCH 19/30] Filter CUJ inference requests before decoding payloads --- tests/README.md | 1 + tests/e2e_cuj/test_cuj4_smart_routing.py | 9 +++++---- tests/test_cuj_evidence.py | 12 ++++++++++++ 3 files changed, 18 insertions(+), 4 deletions(-) diff --git a/tests/README.md b/tests/README.md index 19062c943..75703bf11 100644 --- a/tests/README.md +++ b/tests/README.md @@ -27,6 +27,7 @@ notifications alone cannot substitute for a final parent answer. These offline c establish a live routing pass. Offline CUJ regressions cover Codex delegated-turn completion and exact task-request matching; notifications alone, Claude title requests, and parent continuations do not qualify. +Unrelated and pre-checkpoint request bodies are excluded before JSON decoding. The original smart-routing CUJ runs four fresh sessions: routed and explicit model for both Claude and Codex. One additional test has five `SMART_ROUTER_CONFIG_VERSION` cases. Each case exercises both agents and checks first-prompt routing, orchestrator context in diff --git a/tests/e2e_cuj/test_cuj4_smart_routing.py b/tests/e2e_cuj/test_cuj4_smart_routing.py index 618648fc1..909bda755 100644 --- a/tests/e2e_cuj/test_cuj4_smart_routing.py +++ b/tests/e2e_cuj/test_cuj4_smart_routing.py @@ -89,16 +89,17 @@ def _run_session(session, recorder, agent, task, launch_args): def _task_inference_request(requests, agent, prompt, *, after=0): """Match the task payload, not the expected model or orchestrator context.""" for request in requests: - tools = request.payload.get("tools") if ( request.sequence <= after or request.method != "POST" or request.path != INFERENCE_PATHS[agent] - or not isinstance(tools, list) - or not tools ): continue - entries = request.payload.get("messages" if agent == CLAUDE else "input", []) + payload = request.payload + tools = payload.get("tools") + if not isinstance(tools, list) or not tools: + continue + entries = payload.get("messages" if agent == CLAUDE else "input", []) if isinstance(entries, str): if entries == prompt: return request diff --git a/tests/test_cuj_evidence.py b/tests/test_cuj_evidence.py index a5b362b6f..1e33362eb 100644 --- a/tests/test_cuj_evidence.py +++ b/tests/test_cuj_evidence.py @@ -16,6 +16,7 @@ completed_turn, get_cuj_helper, ) +from tests.e2e_cuj.helpers.tui_request_recorder import RecordedRequest from tests.e2e_cuj.test_cuj4_smart_routing import _task_inference_request from tests.integration.utils.evidence import FileTask, read_jsonl @@ -299,6 +300,17 @@ def _inference_request(agent, sequence, prompt, tools): ) +@pytest.mark.parametrize("agent", [CLAUDE, CODEX]) +@pytest.mark.parametrize( + ("method", "path", "sequence"), + [("GET", None, 2), ("POST", "/models", 2), ("POST", None, 1)], +) +def test_cuj_task_inference_skips_unrelated_empty_bodies(agent, method, path, sequence): + unrelated = RecordedRequest(sequence, method, path or INFERENCE_PATHS[agent], {}, b"") + task = _inference_request(agent, 3, "task prompt", [{"name": "Read"}]) + assert _task_inference_request([unrelated, task], agent, "task prompt", after=1) is task + + def test_cuj_task_inference_selection_uses_exact_tool_prompt(): prompt = "Read input-file.txt using a tool. Reply with only its contents." claude = [ From 424ad936a0faaa7ba1eef9d19f81100f807f3412 Mon Sep 17 00:00:00 2001 From: "lilly.luo" Date: Fri, 9 Oct 2026 06:07:24 +0000 Subject: [PATCH 20/30] Verify Claude thinking-display recovery in smart-routing CUJs --- tests/README.md | 5 +- tests/e2e_cuj/README.md | 5 +- tests/e2e_cuj/helpers/evidence.py | 53 +++++++++++++++++++ tests/e2e_cuj/test_cuj3_models.py | 52 ++++--------------- tests/e2e_cuj/test_cuj4_smart_routing.py | 7 ++- tests/integration/README.md | 6 ++- tests/test_e2e_cuj_helpers.py | 65 +++++++++++++++++++----- 7 files changed, 131 insertions(+), 62 deletions(-) diff --git a/tests/README.md b/tests/README.md index 3c6360ad7..fbd3d0fe8 100644 --- a/tests/README.md +++ b/tests/README.md @@ -22,8 +22,9 @@ CUJs never republish configuration or create a remote reservation. CUJ helper tests also verify that unsupported agent names fail rather than defaulting to Codex. They cover Claude/Codex helper dispatch and rejection of routing decisions without the agent-specific prompt-submission evidence. -CUJ3 also verifies Claude's native recovery from the known thinking-display 400: -the same payload without display must receive 200. +CUJ3 and CUJ4 also verify Claude's native recovery from the known thinking-display 400: +the same payload without display must receive a non-empty 200 for adaptive or enabled thinking. +Offline regressions reject missing/failed retries and changes to the model, prompt, budget, or effort. Native evidence-reader regressions cover Claude's background-agent completion notifications; notifications alone cannot substitute for a final parent answer. These offline checks do not establish a live routing pass. diff --git a/tests/e2e_cuj/README.md b/tests/e2e_cuj/README.md index 524f140c5..8f59d4dce 100644 --- a/tests/e2e_cuj/README.md +++ b/tests/e2e_cuj/README.md @@ -33,6 +33,7 @@ permission prompts. These checks do not prove the gateway's backing destination. establish coverage. Claude 2.1.290 may receive a 400 rejecting `thinking.display: "updates"`; CUJ3 accepts it only if the next task request removes that field, changes nothing else in the payload, and receives a non-empty HTTP 200. Other failures remain test failures. +The shared recovery check covers both adaptive and enabled thinking, preserving any token budget. The test class selects the CUJ3 workspace, `https://dbc-bbdd5508-648e.cloud.databricks.com`. The shared `cuj` fixture supplies its authenticated SDK client and isolated local session; @@ -60,7 +61,7 @@ uv run pytest -c tests/e2e_cuj/pytest.ini --confcutdir=tests/e2e_cuj \ ## Smart-routing CUJ `test_cuj4_smart_routing.py` uses its own read-only workspace with managed smart routing enabled. -The original tests remain unchanged. One additional test has five +It runs the original routed/explicit-model journeys and five `SMART_ROUTER_CONFIG_VERSION` cases using the shared version constants; each launches Claude and Codex once. The independent expectation table is: @@ -80,6 +81,8 @@ Orchestrator presence means its activation context reached the real gateway infe not that an assistant echoed it or a skill merely existed on disk. Task inference must contain the exact task/routed prompt and tools, excluding Claude title requests and parent continuations from child-inference checks. +Parent and child requests use the same verified thinking-display recovery as CUJ3: +only the known 400 followed by an otherwise identical native retry with a non-empty 200 is accepted. Each preset session then explicitly requests one child for a separate hidden-value file task. Assertions require a native child transcript containing the value, the completed parent answer, a correlated spawn-routing decision, and successful child inference on the diff --git a/tests/e2e_cuj/helpers/evidence.py b/tests/e2e_cuj/helpers/evidence.py index 1d757a5db..6a4b90549 100644 --- a/tests/e2e_cuj/helpers/evidence.py +++ b/tests/e2e_cuj/helpers/evidence.py @@ -2,6 +2,7 @@ from __future__ import annotations +import json import re from abc import ABC, abstractmethod from dataclasses import asdict, dataclass @@ -48,6 +49,58 @@ def assert_served(recorder, request, model): assert response.body, "Inference response was empty" +def served_inference_request(recorder, requests, request, agent): + """Require HTTP 200, or Claude's exact native retry without thinking.display.""" + thinking = request.payload.get("thinking", {}) + thinking_type = thinking.get("type") + response = recorder.response_for(request, timeout=240) + known_rejection = False + if ( + agent == CLAUDE + and thinking_type in {"adaptive", "enabled"} + and thinking.get("display") == "updates" + and response.status_code == 400 + ): + try: + error = httpx.Response( + response.status_code, headers=response.headers, content=response.body + ).json() + known_rejection = ( + isinstance(error, dict) + and error.get("error_code") == "BAD_REQUEST" + and json.loads(error.get("message", "")) + == { + "message": f"thinking.{thinking_type}.display: " + "Input should be 'summarized', 'omitted'" + } + ) + except (ValueError, TypeError): + pass + if not known_rejection: + assert_served(recorder, request, request.payload["model"]) + return request + + # Claude 2.1.290 retries once per model/process when Bedrock rejects updates. + # Require the next inference to preserve the full task, model, budget and effort. + # TODO: Remove when Bedrock passthrough accepts thinking-display-updates-2026-08-18. + following = requests[requests.index(request) + 1 :] + retry = next( + ( + candidate + for candidate in following + if candidate.method == request.method and candidate.path == request.path + ), + None, + ) + assert retry is not None, "Thinking display rejection had no retry" + assert retry.payload == { + **request.payload, + "thinking": {key: value for key, value in thinking.items() if key != "display"}, + }, ("Thinking display retry changed more than display", retry.payload) + assert_served(recorder, retry, request.payload["model"]) + return retry + + def claude_file_task(session): """A FileTask naming its absolute path, so Claude reads it without a `find` permission prompt.""" task = FileTask(session) diff --git a/tests/e2e_cuj/test_cuj3_models.py b/tests/e2e_cuj/test_cuj3_models.py index 7eb807210..d427cbf5f 100644 --- a/tests/e2e_cuj/test_cuj3_models.py +++ b/tests/e2e_cuj/test_cuj3_models.py @@ -2,7 +2,6 @@ import json -import httpx import pytest from tests.integration.utils.agents import claude, codex @@ -27,7 +26,13 @@ OTHER_MODEL_SCHEMA, ) from .helpers.constants import CLAUDE, CODEX, INFERENCE_PATHS -from .helpers.evidence import SessionEvidence, assert_served, claude_file_task, message_text +from .helpers.evidence import ( + SessionEvidence, + assert_served, + claude_file_task, + message_text, + served_inference_request, +) from .helpers.terminal import Terminal CUJ_NAME = "CUJ 3 · UC model discovery" @@ -105,23 +110,6 @@ def _request_contains_task(request, agent, task): return False -def _is_thinking_display_rejection(response): - """Recognize the observed gateway rejection, including compressed error bodies.""" - if response.status_code != 400: - return False - try: - error = httpx.Response( - response.status_code, headers=response.headers, content=response.body - ).json() - if not isinstance(error, dict) or error.get("error_code") != "BAD_REQUEST": - return False - return json.loads(error.get("message", "")) == { - "message": "thinking.adaptive.display: Input should be 'summarized', 'omitted'" - } - except (ValueError, TypeError): - return False - - def _assert_inference_evidence(recorder, checkpoint, agent, task, expected): expected_wire_model = claude.discovery_model_id(expected) if agent == CLAUDE else expected requests = recorder.requests_after(checkpoint) @@ -139,29 +127,9 @@ def _assert_inference_evidence(recorder, checkpoint, agent, task, expected): request for request in inference_requests if _request_contains_task(request, agent, task) ] assert task_requests, "No inference request contained the submitted task prompt" - # Claude 2.1.290 sends thinking.display="updates" to custom endpoints; - # 2.1.280 restricted it to Anthropic's first-party base URL. This workspace - # rejects "updates" with 400. Claude retries without it and completes with - # effort="high" unchanged. Accept only this rejection paired with a successful - # retry of the same payload with display removed; all other requests must return 200. - # Live A/B: https://github.com/databricks/unity-gateway/actions/runs/37865757991 - # TODO: Remove this workaround once we add thinking-display-updates-2026-08-18 - # to accepted betas on Bedrock passthrough. - for index, request in enumerate(task_requests): - if ( - agent == CLAUDE - and request.payload.get("thinking") == {"type": "adaptive", "display": "updates"} - and _is_thinking_display_rejection(recorder.response_for(request, timeout=240)) - ): - assert index + 1 < len(task_requests), "Thinking display rejection had no retry" - retry = task_requests[index + 1] - assert retry.payload == {**request.payload, "thinking": {"type": "adaptive"}}, ( - "Thinking display retry changed more than display", - retry.payload, - ) - assert_served(recorder, retry, expected_wire_model) - continue - assert_served(recorder, request, expected_wire_model) + for request in task_requests: + served = served_inference_request(recorder, task_requests, request, agent) + assert_served(recorder, served, expected_wire_model) @pytest.fixture(autouse=True) diff --git a/tests/e2e_cuj/test_cuj4_smart_routing.py b/tests/e2e_cuj/test_cuj4_smart_routing.py index 909bda755..ce383b76e 100644 --- a/tests/e2e_cuj/test_cuj4_smart_routing.py +++ b/tests/e2e_cuj/test_cuj4_smart_routing.py @@ -27,6 +27,7 @@ SessionObservation, canonical_model, claude_file_task, + served_inference_request, ) from .helpers.terminal import Terminal from .helpers.tui_request_recorder import RecordedRequest, RecordedResponse @@ -178,6 +179,7 @@ def run_smart_routing_journeys(cuj) -> SmartRoutingSessionResults: ) route_response = recorder.response_for(route_request) inference_request = _task_inference_request(requests, agent, task.prompt) + inference_request = served_inference_request(recorder, requests, inference_request, agent) no_model_override[agent] = SessionCase( agent=agent, launch_args=(agent,), @@ -194,6 +196,7 @@ def run_smart_routing_journeys(cuj) -> SmartRoutingSessionResults: launch_args = (agent, "--model", overrides[agent]) observation, requests = _run_session(session, recorder, agent, task, launch_args) inference_request = _task_inference_request(requests, agent, task.prompt) + inference_request = served_inference_request(recorder, requests, inference_request, agent) with_model_override[agent] = SessionCase( agent=agent, launch_args=launch_args, @@ -270,7 +273,7 @@ def test_smart_router_config_version( assert len(selections) == 1 expected_model = selections[0]["route_option"]["model"] inference = _task_inference_request(requests, agent, task.prompt) - assert recorder.response_for(inference).status_code == 200 + inference = served_inference_request(recorder, requests, inference, agent) assert canonical_model(inference.payload["model"]) == canonical_model( expected_model ) @@ -333,7 +336,7 @@ def test_smart_router_config_version( route_prompt, after=routes[0].sequence, ) - assert recorder.response_for(inference).status_code == 200 + inference = served_inference_request(recorder, requests, inference, agent) assert canonical_model(inference.payload["model"]) == canonical_model( selections[0]["route_option"]["model"] ) diff --git a/tests/integration/README.md b/tests/integration/README.md index f83a78d3d..23dd3abec 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -10,11 +10,15 @@ CUJ2 adds three separately collected cases for exact published MPS/MCP configura inference, and Claude inference. The config equality also accounts for the workspace's fixture skill name as data; skill download and invocation are not covered. -The dedicated smart-routing CUJ in `../e2e_cuj/test_cuj_smart_routing.py` leaves the original +The dedicated smart-routing CUJ in `../e2e_cuj/test_cuj4_smart_routing.py` leaves the original managed-default and explicit-model cases unchanged. One additional test runs the five supported `SMART_ROUTER_CONFIG_VERSION` values, checking both agents' first-prompt routing, orchestrator context in inference input, and completed explicitly requested routed subagents. This is not automatic orchestrator-delegation coverage. No workspace configuration is modified. +CUJ3 and CUJ4 require a non-empty HTTP 200 for inference. Claude 2.1.290's known +thinking-display 400 is accepted only when the next inference request removes `display`, +preserves every other payload field, and succeeds; adaptive and enabled thinking are covered. +Offline regressions check both CUJs and reject changes to the model, prompt, budget, or effort. Each preset cleans up interactive OS-managed settings with public `ug revert`, even on failure. Offline transcript tests check native Claude background-agent completion evidence; a completion notification alone is not treated as the parent's completed answer. diff --git a/tests/test_e2e_cuj_helpers.py b/tests/test_e2e_cuj_helpers.py index edecb09b9..4a6ccaa76 100644 --- a/tests/test_e2e_cuj_helpers.py +++ b/tests/test_e2e_cuj_helpers.py @@ -20,7 +20,11 @@ INFERENCE_PATHS, CodingAgent, ) -from tests.e2e_cuj.helpers.evidence import assert_models, claude_file_task +from tests.e2e_cuj.helpers.evidence import ( + assert_models, + claude_file_task, + served_inference_request, +) from tests.e2e_cuj.helpers.poll import poll from tests.e2e_cuj.helpers.session import ( MACHINE_WIDE_LEAK, @@ -31,6 +35,7 @@ from tests.e2e_cuj.helpers.terminal import Terminal from tests.e2e_cuj.helpers.workspace import Workspace from tests.e2e_cuj.test_cuj3_models import _assert_inference_evidence, _catalog_display_names +from tests.e2e_cuj.test_cuj4_smart_routing import _task_inference_request def _client(headers): @@ -251,29 +256,40 @@ def test_catalog_display_names_rejects_repeated_pages(pages, message): _catalog_display_names(workspace, CLAUDE, "ug_e2e.models") -@pytest.fixture -def thinking_display_exchange(): +@pytest.fixture(params=["adaptive", "enabled"]) +def thinking_display_exchange(request): model = "ug_e2e.models.claude_sonnet" task = SimpleNamespace(prompt="Read the task file") + thinking = {"type": request.param, "display": "updates"} + if request.param == "enabled": + thinking["budget_tokens"] = 31999 payload = { "model": model, "messages": [{"role": "user", "content": task.prompt}], - "thinking": {"type": "adaptive", "display": "updates"}, + "tools": [{"name": "Read"}], + "thinking": thinking, "output_config": {"effort": "high"}, } requests = [ - SimpleNamespace(method="POST", path=INFERENCE_PATHS[CLAUDE], payload=payload), + SimpleNamespace(sequence=1, method="POST", path=INFERENCE_PATHS[CLAUDE], payload=payload), SimpleNamespace( + sequence=2, method="POST", path=INFERENCE_PATHS[CLAUDE], - payload={**payload, "thinking": {"type": "adaptive"}}, + payload={ + **payload, + "thinking": {key: value for key, value in thinking.items() if key != "display"}, + }, ), ] error = json.dumps( { "error_code": "BAD_REQUEST", "message": json.dumps( - {"message": "thinking.adaptive.display: Input should be 'summarized', 'omitted'"} + { + "message": f"thinking.{request.param}.display: " + "Input should be 'summarized', 'omitted'" + } ), } ).encode() @@ -288,15 +304,20 @@ def thinking_display_exchange(): return recorder, requests, responses, task, model +@pytest.mark.parametrize("contract", ["catalog", "routing"]) @pytest.mark.parametrize("compressed", [False, True]) -def test_catalog_inference_accepts_verified_thinking_display_recovery( - thinking_display_exchange, compressed +def test_cuj_inference_accepts_verified_thinking_display_recovery( + thinking_display_exchange, compressed, contract ): - recorder, _, responses, task, model = thinking_display_exchange + recorder, requests, responses, task, model = thinking_display_exchange if compressed: responses[0].body = gzip.compress(responses[0].body) responses[0].headers = {"content-encoding": "gzip"} - _assert_inference_evidence(recorder, 0, CLAUDE, task, model) + if contract == "catalog": + _assert_inference_evidence(recorder, 0, CLAUDE, task, model) + else: + inference = _task_inference_request(requests, CLAUDE, task.prompt) + assert served_inference_request(recorder, requests, inference, CLAUDE) is requests[1] @pytest.mark.parametrize( @@ -310,17 +331,19 @@ def test_catalog_inference_accepts_verified_thinking_display_recovery( "empty_retry", "changed_model", "changed_effort", + "changed_budget", "changed_prompt", "display_retained", "wrong_display", "codex", ], ) -def test_catalog_inference_rejects_unverified_recovery(thinking_display_exchange, failure): +@pytest.mark.parametrize("contract", ["catalog", "routing"]) +def test_cuj_inference_rejects_unverified_recovery(thinking_display_exchange, failure, contract): recorder, requests, responses, task, model = thinking_display_exchange agent = CLAUDE if failure == "unrelated_400": - responses[0].body = responses[0].body.replace(b"thinking.adaptive.display", b"other.field") + responses[0].body = responses[0].body.replace(b".display", b".other_field") elif failure == "malformed_error": responses[0].body = b"not JSON" elif failure == "server_error": @@ -335,6 +358,8 @@ def test_catalog_inference_rejects_unverified_recovery(thinking_display_exchange requests[1].payload["model"] = "ug_e2e.models.claude_haiku" elif failure == "changed_effort": requests[1].payload["output_config"] = {} + elif failure == "changed_budget": + requests[1].payload["thinking"]["budget_tokens"] = 1000 elif failure == "changed_prompt": requests[1].payload["messages"] = [{"role": "user", "content": "A different task"}] elif failure == "display_retained": @@ -347,7 +372,19 @@ def test_catalog_inference_rejects_unverified_recovery(thinking_display_exchange request.path = INFERENCE_PATHS[CODEX] request.payload["input"] = task.prompt with pytest.raises(AssertionError): - _assert_inference_evidence(recorder, 0, agent, task, model) + if contract == "catalog": + _assert_inference_evidence(recorder, 0, agent, task, model) + else: + inference = _task_inference_request(requests, agent, task.prompt) + served_inference_request(recorder, requests, inference, agent) + + +@pytest.mark.parametrize("agent", [CLAUDE, CODEX]) +def test_served_inference_request_keeps_successful_first_attempt(thinking_display_exchange, agent): + recorder, requests, responses, _, _ = thinking_display_exchange + responses[0].status_code = 200 + responses[0].body = b"successful stream" + assert served_inference_request(recorder, requests, requests[0], agent) is requests[0] def test_assert_models_maps_native_aliases(): From a3ee54881d17c063aabd05305f3d0f6909b49f9d Mon Sep 17 00:00:00 2001 From: "lilly.luo" Date: Fri, 9 Oct 2026 06:08:56 +0000 Subject: [PATCH 21/30] Confirm Claude background-work dialog during TUI exit --- tests/README.md | 3 ++ tests/integration/README.md | 5 ++++ tests/integration/utils/terminal.py | 13 +++++++++ tests/test_e2e_cuj_helpers.py | 44 +++++++++++++++++++++++++++++ 4 files changed, 65 insertions(+) diff --git a/tests/README.md b/tests/README.md index fbd3d0fe8..e6fd07a3c 100644 --- a/tests/README.md +++ b/tests/README.md @@ -20,6 +20,9 @@ offline tests require GET-only API calls and verify config changes fail without The live fixture compares configuration before and after the journey, even on failure. CUJs never republish configuration or create a remote reservation. CUJ helper tests also verify that unsupported agent names fail rather than defaulting to Codex. +Offline PTY checks cover ordinary Claude/Codex exits and Claude 2.1.290's background-work +exit confirmation. The driver confirms the visible "Exit and stop tasks" selection and +rejects other selections; these checks do not establish live agent coverage. They cover Claude/Codex helper dispatch and rejection of routing decisions without the agent-specific prompt-submission evidence. CUJ3 and CUJ4 also verify Claude's native recovery from the known thinking-display 400: diff --git a/tests/integration/README.md b/tests/integration/README.md index 23dd3abec..c597c5d9a 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -38,6 +38,11 @@ or construct ug state files. The normal test suite checks these boundaries. The existing unit tests keep their fixtures. Integration has an independent pytest configuration and uses `--confcutdir` so those fixtures cannot leak in. It is not collected by the default `uv run pytest` command. +After `/exit`, the TUI driver handles Claude 2.1.290's background-work confirmation +by pressing Enter only when "Exit and stop tasks" is visibly selected, then requires +a zero process exit. Offline PTY regressions in `../test_e2e_cuj_helpers.py` cover that +dialog, rejection of other selections, and ordinary Claude/Codex exits; live coverage +still requires running the journeys. `TestChildStdoutLaunch` in `../test_cli.py` covers clean Claude print-mode and Codex exec/app-server stdout, early launch errors, and forwarding through ug's `--`. diff --git a/tests/integration/utils/terminal.py b/tests/integration/utils/terminal.py index 85240dd59..90d1d5d0e 100644 --- a/tests/integration/utils/terminal.py +++ b/tests/integration/utils/terminal.py @@ -402,4 +402,17 @@ def completed(screen): def exit_normally(self): self.submit("/exit") + if self.agent == "claude": + prompt = "Background work is running" + self.wait_for( + lambda text: self.ended or (prompt in text and "Enter to confirm" in text), + "process exit or background-work confirmation", + timeout=30, + ) + if not self.ended: + menu = self.visible.rsplit(prompt, 1)[1] + assert re.search(r"(?m)^\s*[❯›>]\s*1\.\s*Exit and stop tasks\s*$", menu), ( + f"Unrecognized background-work exit selection:\n{self.visible}" + ) + self.send("\r", "confirm Exit and stop tasks") self.finish(timeout=30) diff --git a/tests/test_e2e_cuj_helpers.py b/tests/test_e2e_cuj_helpers.py index 4a6ccaa76..45e656e11 100644 --- a/tests/test_e2e_cuj_helpers.py +++ b/tests/test_e2e_cuj_helpers.py @@ -169,6 +169,50 @@ def approve(screen): assert "allowed" in tui.visible +@requires_pty +@pytest.mark.parametrize("agent", [CLAUDE, CODEX]) +def test_terminal_exit_without_confirmation(tmp_path, agent): + session = _fake_ug(tmp_path, "echo '❯'\nread command\n[ \"$command\" = /exit ]\n") + with Terminal(session, "exits", [], agent=agent) as tui: + tui.exit_normally() + assert tui.ended + assert tui.child.exitstatus == 0 + assert not any(action["reason"] == "confirm Exit and stop tasks" for action in tui.actions) + + +@requires_pty +@pytest.mark.parametrize("selected", ["Exit and stop tasks", "Move to background and exit"]) +def test_terminal_exit_handles_claude_background_work_confirmation(tmp_path, selected): + session = _fake_ug( + tmp_path, + "echo '❯'\n" + "read command\n" + '[ "$command" = /exit ] || exit 1\n' + "echo 'Background work is running'\n" + "echo 'scheduled task · Runs once in 1m · <>'\n" + f"echo '❯ 1. {selected}'\n" + "echo ' 2. Move to background and exit'\n" + "echo ' 3. Stay'\n" + "echo 'Enter to confirm · Esc to cancel'\n" + "read confirmation\n" + '[ -z "$confirmation" ]\n', + ) + with Terminal(session, "confirms-exit", [], agent=CLAUDE) as tui: + if selected == "Exit and stop tasks": + tui.exit_normally() + assert tui.ended + assert tui.child.exitstatus == 0 + assert tui.actions[-1]["reason"] == "confirm Exit and stop tasks" + assert tui.actions[-1]["keys"] == "\r" + else: + with pytest.raises(AssertionError, match="Unrecognized background-work exit selection"): + tui.exit_normally() + assert not tui.ended + assert not any( + action["reason"] == "confirm Exit and stop tasks" for action in tui.actions + ) + + def test_mcp_list_poll_retries_a_failed_probe_until_rows_match(monkeypatch): monkeypatch.setattr(poll_module.time, "sleep", lambda seconds: None) healthy = "\n".join( From 524dba3dbb620579046431c007f80c9203c62297 Mon Sep 17 00:00:00 2001 From: "lilly.luo" Date: Fri, 9 Oct 2026 06:10:02 +0000 Subject: [PATCH 22/30] Wait longer for Claude exit without confirming background-work dialog --- tests/README.md | 3 -- tests/integration/README.md | 8 ++---- tests/integration/utils/terminal.py | 15 +--------- tests/test_e2e_cuj_helpers.py | 44 ----------------------------- 4 files changed, 4 insertions(+), 66 deletions(-) diff --git a/tests/README.md b/tests/README.md index e6fd07a3c..fbd3d0fe8 100644 --- a/tests/README.md +++ b/tests/README.md @@ -20,9 +20,6 @@ offline tests require GET-only API calls and verify config changes fail without The live fixture compares configuration before and after the journey, even on failure. CUJs never republish configuration or create a remote reservation. CUJ helper tests also verify that unsupported agent names fail rather than defaulting to Codex. -Offline PTY checks cover ordinary Claude/Codex exits and Claude 2.1.290's background-work -exit confirmation. The driver confirms the visible "Exit and stop tasks" selection and -rejects other selections; these checks do not establish live agent coverage. They cover Claude/Codex helper dispatch and rejection of routing decisions without the agent-specific prompt-submission evidence. CUJ3 and CUJ4 also verify Claude's native recovery from the known thinking-display 400: diff --git a/tests/integration/README.md b/tests/integration/README.md index c597c5d9a..1cbfbc3c6 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -38,11 +38,9 @@ or construct ug state files. The normal test suite checks these boundaries. The existing unit tests keep their fixtures. Integration has an independent pytest configuration and uses `--confcutdir` so those fixtures cannot leak in. It is not collected by the default `uv run pytest` command. -After `/exit`, the TUI driver handles Claude 2.1.290's background-work confirmation -by pressing Enter only when "Exit and stop tasks" is visibly selected, then requires -a zero process exit. Offline PTY regressions in `../test_e2e_cuj_helpers.py` cover that -dialog, rejection of other selections, and ordinary Claude/Codex exits; live coverage -still requires running the journeys. +After `/exit`, the TUI driver waits up to 120 seconds for Claude to exit normally, +allowing more time for background work. It does not confirm the background-work +dialog. A dialog that remains open still fails on timeout; Codex retains its 30-second wait. `TestChildStdoutLaunch` in `../test_cli.py` covers clean Claude print-mode and Codex exec/app-server stdout, early launch errors, and forwarding through ug's `--`. diff --git a/tests/integration/utils/terminal.py b/tests/integration/utils/terminal.py index 90d1d5d0e..ee37c3649 100644 --- a/tests/integration/utils/terminal.py +++ b/tests/integration/utils/terminal.py @@ -402,17 +402,4 @@ def completed(screen): def exit_normally(self): self.submit("/exit") - if self.agent == "claude": - prompt = "Background work is running" - self.wait_for( - lambda text: self.ended or (prompt in text and "Enter to confirm" in text), - "process exit or background-work confirmation", - timeout=30, - ) - if not self.ended: - menu = self.visible.rsplit(prompt, 1)[1] - assert re.search(r"(?m)^\s*[❯›>]\s*1\.\s*Exit and stop tasks\s*$", menu), ( - f"Unrecognized background-work exit selection:\n{self.visible}" - ) - self.send("\r", "confirm Exit and stop tasks") - self.finish(timeout=30) + self.finish(timeout=120 if self.agent == "claude" else 30) diff --git a/tests/test_e2e_cuj_helpers.py b/tests/test_e2e_cuj_helpers.py index 45e656e11..4a6ccaa76 100644 --- a/tests/test_e2e_cuj_helpers.py +++ b/tests/test_e2e_cuj_helpers.py @@ -169,50 +169,6 @@ def approve(screen): assert "allowed" in tui.visible -@requires_pty -@pytest.mark.parametrize("agent", [CLAUDE, CODEX]) -def test_terminal_exit_without_confirmation(tmp_path, agent): - session = _fake_ug(tmp_path, "echo '❯'\nread command\n[ \"$command\" = /exit ]\n") - with Terminal(session, "exits", [], agent=agent) as tui: - tui.exit_normally() - assert tui.ended - assert tui.child.exitstatus == 0 - assert not any(action["reason"] == "confirm Exit and stop tasks" for action in tui.actions) - - -@requires_pty -@pytest.mark.parametrize("selected", ["Exit and stop tasks", "Move to background and exit"]) -def test_terminal_exit_handles_claude_background_work_confirmation(tmp_path, selected): - session = _fake_ug( - tmp_path, - "echo '❯'\n" - "read command\n" - '[ "$command" = /exit ] || exit 1\n' - "echo 'Background work is running'\n" - "echo 'scheduled task · Runs once in 1m · <>'\n" - f"echo '❯ 1. {selected}'\n" - "echo ' 2. Move to background and exit'\n" - "echo ' 3. Stay'\n" - "echo 'Enter to confirm · Esc to cancel'\n" - "read confirmation\n" - '[ -z "$confirmation" ]\n', - ) - with Terminal(session, "confirms-exit", [], agent=CLAUDE) as tui: - if selected == "Exit and stop tasks": - tui.exit_normally() - assert tui.ended - assert tui.child.exitstatus == 0 - assert tui.actions[-1]["reason"] == "confirm Exit and stop tasks" - assert tui.actions[-1]["keys"] == "\r" - else: - with pytest.raises(AssertionError, match="Unrecognized background-work exit selection"): - tui.exit_normally() - assert not tui.ended - assert not any( - action["reason"] == "confirm Exit and stop tasks" for action in tui.actions - ) - - def test_mcp_list_poll_retries_a_failed_probe_until_rows_match(monkeypatch): monkeypatch.setattr(poll_module.time, "sleep", lambda seconds: None) healthy = "\n".join( From c3cbb47dc5cbfc2a61a573e404cc9f098b338ef7 Mon Sep 17 00:00:00 2001 From: "lilly.luo" Date: Fri, 9 Oct 2026 06:13:52 +0000 Subject: [PATCH 23/30] Wait for Claude background tasks before requesting exit --- tests/README.md | 3 ++ tests/integration/README.md | 7 +-- .../test_ug_smart_routing_hooks.py | 4 +- tests/integration/utils/terminal.py | 17 ++++++- tests/test_e2e_cuj_helpers.py | 48 +++++++++++++++++++ 5 files changed, 74 insertions(+), 5 deletions(-) diff --git a/tests/README.md b/tests/README.md index fbd3d0fe8..1f8e993b2 100644 --- a/tests/README.md +++ b/tests/README.md @@ -20,6 +20,9 @@ offline tests require GET-only API calls and verify config changes fail without The live fixture compares configuration before and after the journey, even on failure. CUJs never republish configuration or create a remote reservation. CUJ helper tests also verify that unsupported agent names fail rather than defaulting to Codex. +Offline PTY checks verify that the Claude background-task wait observes "No tasks currently +running" before sending `/exit`, without stopping tasks or confirming an exit dialog. +The live Claude subagent skill-toggle journey uses this wait after its final calculation. They cover Claude/Codex helper dispatch and rejection of routing decisions without the agent-specific prompt-submission evidence. CUJ3 and CUJ4 also verify Claude's native recovery from the known thinking-display 400: diff --git a/tests/integration/README.md b/tests/integration/README.md index 1cbfbc3c6..8f7b55bc7 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -38,9 +38,10 @@ or construct ug state files. The normal test suite checks these boundaries. The existing unit tests keep their fixtures. Integration has an independent pytest configuration and uses `--confcutdir` so those fixtures cannot leak in. It is not collected by the default `uv run pytest` command. -After `/exit`, the TUI driver waits up to 120 seconds for Claude to exit normally, -allowing more time for background work. It does not confirm the background-work -dialog. A dialog that remains open still fails on timeout; Codex retains its 30-second wait. +The Claude subagent skill-toggle journey opens `/tasks` after its final calculation +and waits up to 180 seconds for "No tasks currently running" before submitting `/exit`. +It neither stops tasks nor confirms an exit dialog. Process exit retains its 30-second +timeout. Offline PTY checks cover this ordering; a live 2.1.290 run is still required. `TestChildStdoutLaunch` in `../test_cli.py` covers clean Claude print-mode and Codex exec/app-server stdout, early launch errors, and forwarding through ug's `--`. diff --git a/tests/integration/test_ug_smart_routing_hooks.py b/tests/integration/test_ug_smart_routing_hooks.py index 3f7505774..1aec53ab9 100644 --- a/tests/integration/test_ug_smart_routing_hooks.py +++ b/tests/integration/test_ug_smart_routing_hooks.py @@ -352,7 +352,8 @@ def test_smart_router_skill_toggles_claude_subagent_routing( confirmation in the native transcript and changes the saved routing controls, even with collapsed terminal output; all three uniquely tagged calculations complete in native child sessions; only the first and third show the subagent-routing banner and produce live gateway - decisions correlated with those children. No first-prompt routing wrapper starts. + decisions correlated with those children. Claude's native task view reports no running + tasks before /exit is submitted. No first-prompt routing wrapper starts. """ session = live_session session.env["TMPDIR"] = str(tmp_path) @@ -393,6 +394,7 @@ def test_smart_router_skill_toggles_claude_subagent_routing( orchestration_enabled=orchestration_enabled, ) _run_calculation(tui, session, "claude", "2+2", "4", routed=True) + tui.wait_for_background_tasks() tui.exit_normally() transcript = "".join(tui.output) assert SMART_ROUTING_BANNER not in transcript, transcript diff --git a/tests/integration/utils/terminal.py b/tests/integration/utils/terminal.py index ee37c3649..11b412ff9 100644 --- a/tests/integration/utils/terminal.py +++ b/tests/integration/utils/terminal.py @@ -400,6 +400,21 @@ def completed(screen): ) task.assert_completed(self.session, self.agent) + def wait_for_background_tasks(self, timeout=180): + """Wait in Claude's native task view without stopping or detaching work.""" + assert self.agent == "claude", self.agent + self.submit("/tasks") + self.wait_for( + lambda text: re.search(r"(?m)^\s*No tasks currently running\s*$", text), + "Claude's task view reporting no running tasks", + timeout=timeout, + ) + self.send("\x1b", "close the completed background-task view") + self.wait_for( + lambda text: "No tasks currently running" not in text, + "the prompt after closing the background-task view", + ) + def exit_normally(self): self.submit("/exit") - self.finish(timeout=120 if self.agent == "claude" else 30) + self.finish(timeout=30) diff --git a/tests/test_e2e_cuj_helpers.py b/tests/test_e2e_cuj_helpers.py index 4a6ccaa76..58a3d156d 100644 --- a/tests/test_e2e_cuj_helpers.py +++ b/tests/test_e2e_cuj_helpers.py @@ -169,6 +169,54 @@ def approve(screen): assert "allowed" in tui.visible +@requires_pty +@pytest.mark.parametrize("complete", [True, False]) +def test_claude_waits_for_background_task_completion_before_exit(tmp_path, complete): + """An offline terminal fixture requires /tasks, completion, Escape, then /exit.""" + script = tmp_path / "ug" + script.write_text( + f"""#!{sys.executable} +import sys +import termios +import time +import tty + +print("❯", flush=True) +assert sys.stdin.readline().strip() == "/tasks" +print("Background tasks: scheduled task · Runs once in 1m", flush=True) +if not {complete!r}: + time.sleep(30) + raise SystemExit(1) +time.sleep(0.8) +previous = termios.tcgetattr(sys.stdin.fileno()) +tty.setraw(sys.stdin.fileno()) +print("\\x1b[2J\\x1b[HNo tasks currently running", flush=True) +assert sys.stdin.read(1) == "\\x1b" +termios.tcsetattr(sys.stdin.fileno(), termios.TCSANOW, previous) +print("\\x1b[2J\\x1b[H❯", flush=True) +assert sys.stdin.readline().strip() == "/exit" +""" + ) + script.chmod(0o755) + session = UserSession(tmp_path, script, tmp_path / "artifacts", "token") + with Terminal(session, "waits-before-exit", [], agent=CLAUDE) as tui: + if complete: + tui.wait_for_background_tasks(timeout=5) + tui.exit_normally() + assert tui.child.exitstatus == 0 + close = next( + action + for action in tui.actions + if action["reason"] == "close the completed background-task view" + ) + assert "No tasks currently running" in close["screen_before"] + else: + with pytest.raises(AssertionError, match="task view reporting no running tasks"): + tui.wait_for_background_tasks(timeout=0.3) + assert not tui.ended + assert not any("/exit" in action["keys"] for action in tui.actions) + + def test_mcp_list_poll_retries_a_failed_probe_until_rows_match(monkeypatch): monkeypatch.setattr(poll_module.time, "sleep", lambda seconds: None) healthy = "\n".join( From 2c4e6bd720ea24d6f578283586a93872980f473e Mon Sep 17 00:00:00 2001 From: "lilly.luo" Date: Fri, 9 Oct 2026 06:24:36 +0000 Subject: [PATCH 24/30] Verify chained Claude native compatibility retries in CUJs --- tests/README.md | 1 + tests/e2e_cuj/README.md | 6 +- tests/e2e_cuj/helpers/evidence.py | 107 ++++++++++++++++-------------- tests/integration/README.md | 2 + tests/test_e2e_cuj_helpers.py | 106 +++++++++++++++++++++++++++++ 5 files changed, 172 insertions(+), 50 deletions(-) diff --git a/tests/README.md b/tests/README.md index 1f8e993b2..0d5673c93 100644 --- a/tests/README.md +++ b/tests/README.md @@ -27,6 +27,7 @@ They cover Claude/Codex helper dispatch and rejection of routing decisions witho the agent-specific prompt-submission evidence. CUJ3 and CUJ4 also verify Claude's native recovery from the known thinking-display 400: the same payload without display must receive a non-empty 200 for adaptive or enabled thinking. +If the next attempt rejects `safeguards`, only its native removal is accepted before the final 200. Offline regressions reject missing/failed retries and changes to the model, prompt, budget, or effort. Native evidence-reader regressions cover Claude's background-agent completion notifications; notifications alone cannot substitute for a final parent answer. These offline checks do not diff --git a/tests/e2e_cuj/README.md b/tests/e2e_cuj/README.md index 8f59d4dce..1d3c930ac 100644 --- a/tests/e2e_cuj/README.md +++ b/tests/e2e_cuj/README.md @@ -34,6 +34,9 @@ establish coverage. Claude 2.1.290 may receive a 400 rejecting `thinking.display CUJ3 accepts it only if the next task request removes that field, changes nothing else in the payload, and receives a non-empty HTTP 200. Other failures remain test failures. The shared recovery check covers both adaptive and enabled thinking, preserving any token budget. +Claude may then receive `safeguards: Extra inputs are not permitted`. Recovery must remove +only that field in the next native attempt and end with a non-empty HTTP 200. The checks +inspect recorded traffic without modifying or replaying requests. The test class selects the CUJ3 workspace, `https://dbc-bbdd5508-648e.cloud.databricks.com`. The shared `cuj` fixture supplies its authenticated SDK client and isolated local session; @@ -82,7 +85,8 @@ not that an assistant echoed it or a skill merely existed on disk. Task inference must contain the exact task/routed prompt and tools, excluding Claude title requests and parent continuations from child-inference checks. Parent and child requests use the same verified thinking-display recovery as CUJ3: -only the known 400 followed by an otherwise identical native retry with a non-empty 200 is accepted. +only known display/safeguards rejections followed by native removal of the rejected field +and a final non-empty 200 are accepted. Model, prompt, budget, and effort must remain unchanged. Each preset session then explicitly requests one child for a separate hidden-value file task. Assertions require a native child transcript containing the value, the completed parent answer, a correlated spawn-routing decision, and successful child inference on the diff --git a/tests/e2e_cuj/helpers/evidence.py b/tests/e2e_cuj/helpers/evidence.py index 6a4b90549..e63c7a537 100644 --- a/tests/e2e_cuj/helpers/evidence.py +++ b/tests/e2e_cuj/helpers/evidence.py @@ -50,55 +50,64 @@ def assert_served(recorder, request, model): def served_inference_request(recorder, requests, request, agent): - """Require HTTP 200, or Claude's exact native retry without thinking.display.""" - thinking = request.payload.get("thinking", {}) - thinking_type = thinking.get("type") - response = recorder.response_for(request, timeout=240) - known_rejection = False - if ( - agent == CLAUDE - and thinking_type in {"adaptive", "enabled"} - and thinking.get("display") == "updates" - and response.status_code == 400 - ): - try: - error = httpx.Response( - response.status_code, headers=response.headers, content=response.body - ).json() - known_rejection = ( - isinstance(error, dict) - and error.get("error_code") == "BAD_REQUEST" - and json.loads(error.get("message", "")) - == { - "message": f"thinking.{thinking_type}.display: " - "Input should be 'summarized', 'omitted'" - } - ) - except (ValueError, TypeError): - pass - if not known_rejection: - assert_served(recorder, request, request.payload["model"]) - return request - - # Claude 2.1.290 retries once per model/process when Bedrock rejects updates. - # Require the next inference to preserve the full task, model, budget and effort. - # TODO: Remove when Bedrock passthrough accepts thinking-display-updates-2026-08-18. - following = requests[requests.index(request) + 1 :] - retry = next( - ( - candidate - for candidate in following - if candidate.method == request.method and candidate.path == request.path - ), - None, - ) - assert retry is not None, "Thinking display rejection had no retry" - assert retry.payload == { - **request.payload, - "thinking": {key: value for key, value in thinking.items() if key != "display"}, - }, ("Thinking display retry changed more than display", retry.payload) - assert_served(recorder, retry, request.payload["model"]) - return retry + """Require HTTP 200 or verified native Claude compatibility retries.""" + model = request.payload["model"] + # Claude 2.1.290 may remove display, then safeguards after separate Bedrock + # validation errors. Inspect recorded traffic only; never replay or edit it. + # Each field may disappear once, and the full task/model/budget/effort must survive. + for _ in range(3): + payload = request.payload + thinking = payload.get("thinking", {}) + thinking_type = thinking.get("type") + response = recorder.response_for(request, timeout=240) + error_message = None + if agent == CLAUDE and response.status_code == 400: + try: + error = httpx.Response( + response.status_code, headers=response.headers, content=response.body + ).json() + if isinstance(error, dict) and error.get("error_code") == "BAD_REQUEST": + detail = json.loads(error.get("message", "")) + if isinstance(detail, dict) and set(detail) == {"message"}: + error_message = detail["message"] + except (ValueError, TypeError): + pass + expected_retry = None + if ( + thinking_type in {"adaptive", "enabled"} + and thinking.get("display") == "updates" + and error_message + == f"thinking.{thinking_type}.display: Input should be 'summarized', 'omitted'" + ): + expected_retry = { + **payload, + "thinking": {key: value for key, value in thinking.items() if key != "display"}, + } + elif ( + "safeguards" in payload + and error_message == "safeguards: Extra inputs are not permitted" + ): + expected_retry = {key: value for key, value in payload.items() if key != "safeguards"} + if expected_retry is None: + assert_served(recorder, request, model) + return request + following = requests[requests.index(request) + 1 :] + retry = next( + ( + candidate + for candidate in following + if candidate.method == request.method and candidate.path == request.path + ), + None, + ) + assert retry is not None, "Claude compatibility rejection had no retry" + assert retry.payload == expected_retry, ( + "Claude compatibility retry changed more than the rejected field", + error_message, + retry.payload, + ) + request = retry + raise AssertionError("Claude compatibility retries did not reach a successful response") def claude_file_task(session): diff --git a/tests/integration/README.md b/tests/integration/README.md index 8f7b55bc7..f2b33531e 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -18,6 +18,8 @@ This is not automatic orchestrator-delegation coverage. No workspace configurati CUJ3 and CUJ4 require a non-empty HTTP 200 for inference. Claude 2.1.290's known thinking-display 400 is accepted only when the next inference request removes `display`, preserves every other payload field, and succeeds; adaptive and enabled thinking are covered. +The native retry may also encounter `safeguards: Extra inputs are not permitted`; that +requires the next recorded attempt to remove only `safeguards` and reach a non-empty 200. Offline regressions check both CUJs and reject changes to the model, prompt, budget, or effort. Each preset cleans up interactive OS-managed settings with public `ug revert`, even on failure. Offline transcript tests check native Claude background-agent completion evidence; diff --git a/tests/test_e2e_cuj_helpers.py b/tests/test_e2e_cuj_helpers.py index 58a3d156d..9e62835e8 100644 --- a/tests/test_e2e_cuj_helpers.py +++ b/tests/test_e2e_cuj_helpers.py @@ -435,6 +435,112 @@ def test_served_inference_request_keeps_successful_first_attempt(thinking_displa assert served_inference_request(recorder, requests, requests[0], agent) is requests[0] +@pytest.fixture +def safeguards_exchange(thinking_display_exchange): + recorder, requests, responses, task, model = thinking_display_exchange + requests[0].payload["safeguards"] = {"enabled": True} + requests[1].payload["safeguards"] = {"enabled": True} + responses[1].status_code = 400 + responses[1].body = json.dumps( + { + "error_code": "BAD_REQUEST", + "message": json.dumps({"message": "safeguards: Extra inputs are not permitted"}), + } + ).encode() + requests.append( + SimpleNamespace( + sequence=3, + method="POST", + path=INFERENCE_PATHS[CLAUDE], + payload={ + key: value for key, value in requests[1].payload.items() if key != "safeguards" + }, + ) + ) + responses.append(SimpleNamespace(status_code=200, headers={}, body=b"successful stream")) + return recorder, requests, responses, task, model + + +@pytest.mark.parametrize("contract", ["catalog", "routing"]) +@pytest.mark.parametrize("compressed", [False, True]) +def test_cuj_inference_accepts_chained_native_compatibility_recovery( + safeguards_exchange, contract, compressed +): + recorder, requests, responses, task, model = safeguards_exchange + if compressed: + for response in responses[:2]: + response.body = gzip.compress(response.body) + response.headers = {"content-encoding": "gzip"} + if contract == "catalog": + _assert_inference_evidence(recorder, 0, CLAUDE, task, model) + else: + inference = _task_inference_request(requests, CLAUDE, task.prompt) + assert served_inference_request(recorder, requests, inference, CLAUDE) is requests[2] + + +def test_served_inference_accepts_safeguards_recovery_without_display_rejection( + safeguards_exchange, +): + recorder, requests, _, _, _ = safeguards_exchange + assert served_inference_request(recorder, requests, requests[1], CLAUDE) is requests[2] + + +@pytest.mark.parametrize("contract", ["catalog", "routing"]) +@pytest.mark.parametrize( + "failure", + [ + "missing_retry", + "failed_retry", + "empty_retry", + "changed_model", + "changed_effort", + "changed_budget", + "changed_prompt", + "safeguards_retained", + "unrelated_error", + "missing_safeguards", + "codex", + ], +) +def test_cuj_inference_rejects_unverified_safeguards_recovery( + safeguards_exchange, contract, failure +): + recorder, requests, responses, task, model = safeguards_exchange + agent = CLAUDE + if failure == "missing_retry": + requests.pop() + elif failure == "failed_retry": + responses[2].status_code = 400 + elif failure == "empty_retry": + responses[2].body = b"" + elif failure == "changed_model": + requests[2].payload["model"] = "different-model" + elif failure == "changed_effort": + requests[2].payload["output_config"] = {} + elif failure == "changed_budget": + requests[2].payload["thinking"] = {"type": "enabled", "budget_tokens": 1000} + elif failure == "changed_prompt": + requests[2].payload["messages"] = [{"role": "user", "content": "different task"}] + elif failure == "safeguards_retained": + requests[2].payload["safeguards"] = requests[1].payload["safeguards"] + elif failure == "unrelated_error": + responses[1].body = responses[1].body.replace(b"safeguards:", b"unrelated:") + elif failure == "missing_safeguards": + for request in requests: + request.payload.pop("safeguards", None) + elif failure == "codex": + agent = CODEX + for request in requests: + request.path = INFERENCE_PATHS[CODEX] + request.payload["input"] = task.prompt + with pytest.raises(AssertionError): + if contract == "catalog": + _assert_inference_evidence(recorder, 0, agent, task, model) + else: + inference = _task_inference_request(requests, agent, task.prompt) + served_inference_request(recorder, requests, inference, agent) + + def test_assert_models_maps_native_aliases(): alias = "anthropic.claude-haiku-4-5-20251001-v1:0" assert_models([alias, CLAUDE_HAIKU_MODEL], CLAUDE_HAIKU_MODEL) From 913ce4efd367e4b1417a16d8c33590f6cdfb988c Mon Sep 17 00:00:00 2001 From: "lilly.luo" Date: Fri, 9 Oct 2026 06:56:58 +0000 Subject: [PATCH 25/30] Acknowledge Claude auto-mode billing notice in CUJ TUI waits Unrouted Claude sessions stay in auto mode, and their classifier requests through the recording proxy raise an informational modal that blocked the parent turn until the 240s timeout. Continue past that exact notice once. Co-authored-by: Isaac --- tests/e2e_cuj/README.md | 4 ++++ tests/e2e_cuj/helpers/terminal.py | 19 +++++++++++++++++++ 2 files changed, 23 insertions(+) diff --git a/tests/e2e_cuj/README.md b/tests/e2e_cuj/README.md index 1d3c930ac..d1567a6a6 100644 --- a/tests/e2e_cuj/README.md +++ b/tests/e2e_cuj/README.md @@ -94,6 +94,10 @@ router's selected model. This tests requested delegation, not automatic orchestr Codex requires a matching completed turn and final parent answer, not just child notifications. Offline evidence regressions do not establish a live CUJ pass. Selector cases have separate TUI artifact names, and the session environment is restored afterward. +Claude sessions start in auto mode; unless a routed first prompt switches the session to Haiku +(which leaves it in manual mode), its classifier requests through the recording proxy trigger +Claude's informational auto-mode classifier billing notice over the transcript. The CUJ terminal +acknowledges that exact notice once with Enter (continue); any other dialog still fails. Each preset also runs public `ug revert` in cleanup, including after a failed assertion, so interactive launches' OS-managed settings cannot contaminate the next preset's configuration. The existing Claude explicit-model precedence case remains skipped; the routing-disabled case diff --git a/tests/e2e_cuj/helpers/terminal.py b/tests/e2e_cuj/helpers/terminal.py index 3ef58715a..678bbbe14 100644 --- a/tests/e2e_cuj/helpers/terminal.py +++ b/tests/e2e_cuj/helpers/terminal.py @@ -6,6 +6,14 @@ from .constants import CLAUDE, CODEX +def _shows_auto_mode_billing_notice(screen): + return ( + "We're changing auto mode to no longer charge for classifier requests" in screen + and "Nothing breaks: auto mode keeps working" in screen + and "Enter to continue · Esc to cancel" in screen + ) + + class Terminal(AgentTerminal): def __init__(self, session, name, args, *, agent=None): """Run `ug `; bare `ug` (empty args) names the agent it is expected to launch.""" @@ -26,14 +34,25 @@ def wait_until( """Wait for `done()`, failing on API errors or any `rejected` permission prompt. `on_screen(screen)` may answer an expected dialog; it returns True when it sent keys. + Claude's one-time auto-mode classifier billing notice is acknowledged with Enter. """ + notice_acknowledged = False def completed(screen): + nonlocal notice_acknowledged assert_no_terminal_api_error(screen) # Do not use wait_for_task's optional tool-permission approval. assert not any(prompt in screen for prompt in rejected), ( "Unexpected permission request; inspect the actual command:\n" + screen ) + if _shows_auto_mode_billing_notice(screen): + # Auto-mode classifier requests through the recording proxy raise this + # informational modal over the transcript. Continue keeps behavior unchanged; + # acknowledge it once. + if not notice_acknowledged: + self.send("\r", "acknowledge auto-mode classifier billing notice") + notice_acknowledged = True + return False if on_screen is not None and on_screen(screen): return False return done() From dfc206961bc67ae655a32ae63a1b091dfdcf6fef Mon Sep 17 00:00:00 2001 From: "lilly.luo" Date: Fri, 9 Oct 2026 16:47:31 +0000 Subject: [PATCH 26/30] Fix Claude smart-routing CUJ completion evidence Wait for billing-notice input readiness and confirm dismissal instead of leaving a visible notice blocked after the first Enter. Correlate native peer hand-backs with successful Agent spawns so the final parent answer remains in the requested turn. Accept only Claude child requests' single transport newline after the routing checkpoint. Add PTY and negative evidence regressions and update coverage docs. --- tests/README.md | 6 ++ tests/e2e_cuj/README.md | 9 ++- tests/e2e_cuj/helpers/evidence.py | 37 ++++++++++++- tests/e2e_cuj/helpers/terminal.py | 21 ++++--- tests/e2e_cuj/test_cuj4_smart_routing.py | 12 ++-- tests/integration/README.md | 6 ++ tests/test_cuj_evidence.py | 70 ++++++++++++++++++++++++ tests/test_e2e_cuj_helpers.py | 41 ++++++++++++++ 8 files changed, 187 insertions(+), 15 deletions(-) diff --git a/tests/README.md b/tests/README.md index 0d5673c93..23e32a691 100644 --- a/tests/README.md +++ b/tests/README.md @@ -32,6 +32,12 @@ Offline regressions reject missing/failed retries and changes to the model, prom Native evidence-reader regressions cover Claude's background-agent completion notifications; notifications alone cannot substitute for a final parent answer. These offline checks do not establish a live routing pass. +Claude peer hand-backs are correlated with successful Agent spawns in the same parent turn; +unknown senders, mismatched sessions, failed spawns, and reports without a final parent answer +are rejected. Offline PTY checks cover billing-notice input readiness, observed dismissal, +and a later notice for child work. +Child HTTP matching accepts Claude's single appended transport newline after the routing +checkpoint; additional text, whitespace, and Codex prompt changes remain rejected. Offline CUJ regressions cover Codex delegated-turn completion and exact task-request matching; notifications alone, Claude title requests, and parent continuations do not qualify. Unrelated and pre-checkpoint request bodies are excluded before JSON decoding. diff --git a/tests/e2e_cuj/README.md b/tests/e2e_cuj/README.md index d1567a6a6..90fd648a6 100644 --- a/tests/e2e_cuj/README.md +++ b/tests/e2e_cuj/README.md @@ -83,7 +83,8 @@ the routed or configured default model, completed native file-task evidence, and Orchestrator presence means its activation context reached the real gateway inference input, not that an assistant echoed it or a skill merely existed on disk. Task inference must contain the exact task/routed prompt and tools, excluding Claude title -requests and parent continuations from child-inference checks. +requests and parent continuations from child-inference checks. Only Claude child requests +after the routing checkpoint may append one transport newline to that exact prompt. Parent and child requests use the same verified thinking-display recovery as CUJ3: only known display/safeguards rejections followed by native removal of the rejected field and a final non-empty 200 are accepted. Model, prompt, budget, and effort must remain unchanged. @@ -97,7 +98,11 @@ Selector cases have separate TUI artifact names, and the session environment is Claude sessions start in auto mode; unless a routed first prompt switches the session to Haiku (which leaves it in manual mode), its classifier requests through the recording proxy trigger Claude's informational auto-mode classifier billing notice over the transcript. The CUJ terminal -acknowledges that exact notice once with Enter (continue); any other dialog still fails. +waits for that exact notice to render stably, presses Enter (continue), and observes dismissal +before continuing. Later occurrences are handled the same way; any other dialog still fails. +Claude's native peer hand-back remains part of the parent's turn only when its sender matches +a successful Agent spawn in that turn. The hand-back alone cannot satisfy completion: the +parent must still produce its own final answer with the hidden value. Each preset also runs public `ug revert` in cleanup, including after a failed assertion, so interactive launches' OS-managed settings cannot contaminate the next preset's configuration. The existing Claude explicit-model precedence case remains skipped; the routing-disabled case diff --git a/tests/e2e_cuj/helpers/evidence.py b/tests/e2e_cuj/helpers/evidence.py index e63c7a537..4371918f4 100644 --- a/tests/e2e_cuj/helpers/evidence.py +++ b/tests/e2e_cuj/helpers/evidence.py @@ -188,17 +188,52 @@ def _completed_turn(records, task): first = records[prompts[0]] session_id = first.get("sessionId") responses = [] + spawn_tools = set() + spawned_children = set() for row in records[prompts[0] + 1 :]: msg = row.get("message", {}) + origin = row.get("origin", {}) + child_report = ( + origin.get("kind") == "peer" + and origin.get("handback") is True + and origin.get("from") in spawned_children + and origin.get("senderTaskId") == origin.get("from") + and row.get("sessionId") == session_id + ) if ( row.get("type") == "user" and isinstance(msg.get("content"), str) - and row.get("origin", {}).get("kind") != "task-notification" + and origin.get("kind") != "task-notification" + and not child_report ): break + if row.get("type") == "user" and isinstance(msg.get("content"), list): + result = row.get("toolUseResult", {}) + if ( + row.get("sessionId") == session_id + and result.get("status") == "async_launched" + and result.get("isAsync") is True + and isinstance(result.get("agentId"), str) + and result["agentId"] + and any( + part.get("type") == "tool_result" + and part.get("tool_use_id") in spawn_tools + and not part.get("is_error") + for part in msg["content"] + ) + ): + spawned_children.add(result["agentId"]) if row.get("type") == "assistant": assert row.get("sessionId") == session_id and session_id responses.append(msg) + spawn_tools.update( + part["id"] + for part in msg.get("content", []) + if part.get("type") == "tool_use" + and part.get("name") == "Agent" + and isinstance(part.get("id"), str) + and part["id"] + ) if not responses: return None last = responses[-1] diff --git a/tests/e2e_cuj/helpers/terminal.py b/tests/e2e_cuj/helpers/terminal.py index 678bbbe14..044b8c8ae 100644 --- a/tests/e2e_cuj/helpers/terminal.py +++ b/tests/e2e_cuj/helpers/terminal.py @@ -34,12 +34,10 @@ def wait_until( """Wait for `done()`, failing on API errors or any `rejected` permission prompt. `on_screen(screen)` may answer an expected dialog; it returns True when it sent keys. - Claude's one-time auto-mode classifier billing notice is acknowledged with Enter. + Claude's auto-mode classifier billing notice is acknowledged with Enter. """ - notice_acknowledged = False def completed(screen): - nonlocal notice_acknowledged assert_no_terminal_api_error(screen) # Do not use wait_for_task's optional tool-permission approval. assert not any(prompt in screen for prompt in rejected), ( @@ -47,11 +45,18 @@ def completed(screen): ) if _shows_auto_mode_billing_notice(screen): # Auto-mode classifier requests through the recording proxy raise this - # informational modal over the transcript. Continue keeps behavior unchanged; - # acknowledge it once. - if not notice_acknowledged: - self.send("\r", "acknowledge auto-mode classifier billing notice") - notice_acknowledged = True + # informational modal over the transcript. Wait for its input handler + # before pressing Enter, then observe dismissal before continuing. + self.wait_for( + _shows_auto_mode_billing_notice, + "the auto-mode classifier billing notice", + stable_for=0.5, + ) + self.send("\r", "acknowledge auto-mode classifier billing notice") + self.wait_for( + lambda text: not _shows_auto_mode_billing_notice(text), + "dismissal of the auto-mode classifier billing notice", + ) return False if on_screen is not None and on_screen(screen): return False diff --git a/tests/e2e_cuj/test_cuj4_smart_routing.py b/tests/e2e_cuj/test_cuj4_smart_routing.py index ce383b76e..235cf599e 100644 --- a/tests/e2e_cuj/test_cuj4_smart_routing.py +++ b/tests/e2e_cuj/test_cuj4_smart_routing.py @@ -88,7 +88,10 @@ def _run_session(session, recorder, agent, task, launch_args): def _task_inference_request(requests, agent, prompt, *, after=0): - """Match the task payload, not the expected model or orchestrator context.""" + """Match the task payload, allowing Claude's single child-prompt newline.""" + prompts = {prompt} + if agent == CLAUDE and after: + prompts.add(prompt + "\n") for request in requests: if ( request.sequence <= after @@ -102,7 +105,7 @@ def _task_inference_request(requests, agent, prompt, *, after=0): continue entries = payload.get("messages" if agent == CLAUDE else "input", []) if isinstance(entries, str): - if entries == prompt: + if entries in prompts: return request continue if not isinstance(entries, list): @@ -112,12 +115,13 @@ def _task_inference_request(requests, agent, prompt, *, after=0): continue content = entry.get("content", []) if isinstance(content, str): - if content == prompt: + if content in prompts: return request elif isinstance(content, list) and any( isinstance(part, dict) and part.get("type") in {"text", "input_text"} - and part.get("text") == prompt + and isinstance(part.get("text"), str) + and part.get("text") in prompts for part in content ): return request diff --git a/tests/integration/README.md b/tests/integration/README.md index f2b33531e..6bfd036bb 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -24,6 +24,12 @@ Offline regressions check both CUJs and reject changes to the model, prompt, bud Each preset cleans up interactive OS-managed settings with public `ug revert`, even on failure. Offline transcript tests check native Claude background-agent completion evidence; a completion notification alone is not treated as the parent's completed answer. +Native peer hand-backs also remain in the parent turn when correlated with a successful +Agent spawn; the parent must still answer. Offline negative controls reject unrelated reports. +The CUJ terminal waits for stable billing-notice rendering before Enter and observes dismissal; +an offline PTY regression also exercises a later notice during child work. +Claude child HTTP prompts may carry one appended newline after the route-selection checkpoint; +offline cases reject any other prompt changes and retain exact Codex matching. Offline CUJ cases also cover Codex delegated-turn completion and task-request matching; see `../e2e_cuj/README.md` for the evidence requirements. diff --git a/tests/test_cuj_evidence.py b/tests/test_cuj_evidence.py index 1e33362eb..a60d22761 100644 --- a/tests/test_cuj_evidence.py +++ b/tests/test_cuj_evidence.py @@ -181,6 +181,63 @@ def test_cuj_evidence_claude_real_user_message_ends_parent_turn(): assert completed_turn(CLAUDE, [prompt, user, answer], task) is None +@pytest.mark.parametrize( + "failure", + [ + None, + "no_answer", + "unknown_child", + "wrong_sender", + "wrong_session", + "not_handback", + "failed_spawn", + "unrelated_tool", + ], +) +def test_cuj_evidence_claude_spawned_child_handback_keeps_parent_turn(failure): + task = SimpleNamespace(prompt="Delegate reading the file.", value="hidden-value") + prompt, answer = records(CLAUDE, task, "system.ai.claude-sonnet-4-6") + delegation = copy.deepcopy(answer) + delegation["message"].update( + id="delegation", + stop_reason="tool_use", + content=[{"type": "tool_use", "id": "spawn", "name": "Agent", "input": {}}], + ) + spawned = { + "type": "user", + "sessionId": "session", + "message": {"content": [{"type": "tool_result", "tool_use_id": "spawn"}]}, + "toolUseResult": {"status": "async_launched", "isAsync": True, "agentId": "child"}, + } + handback = { + "type": "user", + "sessionId": "session", + "origin": {"kind": "peer", "handback": True, "from": "child", "senderTaskId": "child"}, + "message": {"content": f"Another Claude session sent a message: {task.value}"}, + } + if failure == "unknown_child": + handback["origin"].update({"from": "other", "senderTaskId": "other"}) + elif failure == "wrong_sender": + handback["origin"]["senderTaskId"] = "other" + elif failure == "wrong_session": + handback["sessionId"] = "other" + elif failure == "not_handback": + handback["origin"]["handback"] = False + elif failure == "failed_spawn": + spawned["message"]["content"][0]["is_error"] = True + elif failure == "unrelated_tool": + delegation["message"]["content"][0]["name"] = "Read" + rows = [prompt, delegation, spawned, handback] + if failure != "no_answer": + rows.append(answer) + turn = completed_turn(CLAUDE, rows, task) + if failure is None: + assert turn is not None + assert (turn.turn_id, turn.answer) == ("response", task.value) + else: + assert turn is None + + @pytest.mark.parametrize("agent", [CLAUDE, CODEX]) @pytest.mark.parametrize( "failure", @@ -329,6 +386,19 @@ def test_cuj_task_inference_selection_uses_exact_tool_prompt(): _task_inference_request([parent, child], CODEX, child_prompt, after=5) +@pytest.mark.parametrize("agent", [CLAUDE, CODEX]) +@pytest.mark.parametrize("after", [0, 1]) +@pytest.mark.parametrize("suffix", ["\n", "\n\n", " ", "\nDifferent task."]) +def test_cuj_task_inference_only_allows_claude_child_transport_newline(agent, after, suffix): + prompt = "Read input-file.txt using a tool and return its contents." + request = _inference_request(agent, 2, prompt + suffix, [{"name": "Read"}]) + if agent == CLAUDE and after and suffix == "\n": + assert _task_inference_request([request], agent, prompt, after=after) is request + else: + with pytest.raises(AssertionError, match="No tool-capable"): + _task_inference_request([request], agent, prompt, after=after) + + def test_cuj_evidence_ignores_existing_session(tmp_path): task = FileTask(SimpleNamespace(cwd=tmp_path)) first = SessionEvidence(tmp_path, CLAUDE) diff --git a/tests/test_e2e_cuj_helpers.py b/tests/test_e2e_cuj_helpers.py index 9e62835e8..62892fcfd 100644 --- a/tests/test_e2e_cuj_helpers.py +++ b/tests/test_e2e_cuj_helpers.py @@ -169,6 +169,47 @@ def approve(screen): assert "allowed" in tui.visible +@requires_pty +def test_wait_until_acknowledges_ready_billing_notices_and_observes_dismissal(tmp_path): + """A rendered notice can precede its input handler and appear again for a child.""" + done = tmp_path / "done" + script = tmp_path / "ug" + script.write_text( + f"""#!{sys.executable} +import sys +import termios +import time +import tty +from pathlib import Path + +tty.setraw(sys.stdin.fileno()) +for _ in range(2): + print("\\x1b[2J\\x1b[HWe're changing auto mode to no longer charge for classifier requests", flush=True) + print("Nothing breaks: auto mode keeps working", flush=True) + print("Enter to continue · Esc to cancel", flush=True) + time.sleep(0.2) + termios.tcflush(sys.stdin.fileno(), termios.TCIFLUSH) + assert sys.stdin.read(1) == "\\r" + print("\\x1b[2J\\x1b[HWorking", flush=True) + time.sleep(0.8) +Path({str(done)!r}).touch() +time.sleep(30) +""" + ) + script.chmod(0o755) + session = UserSession(tmp_path, script, tmp_path / "artifacts", "token") + with Terminal(session, "billing-notices", [], agent=CLAUDE) as tui: + tui.wait_until(done.exists, "completed work after both billing notices", timeout=10) + acknowledgements = [ + action + for action in tui.actions + if action["reason"] == "acknowledge auto-mode classifier billing notice" + ] + assert len(acknowledgements) == 2 + assert all(action["keys"] == "\r" for action in acknowledgements) + assert "Enter to continue" not in tui.visible + + @requires_pty @pytest.mark.parametrize("complete", [True, False]) def test_claude_waits_for_background_task_completion_before_exit(tmp_path, complete): From 52dc9126ef940f87ee09d84e0485d7cbcf41edf2 Mon Sep 17 00:00:00 2001 From: "lilly.luo" Date: Fri, 9 Oct 2026 17:45:48 +0000 Subject: [PATCH 27/30] Recognize completed Claude background tasks before exit Claude 2.1.290 retains finished children in a Completed task list instead of showing the empty task view. Accept only a fully rendered native menu whose completed count matches its checkmarked done rows, then observe menu dismissal before submitting /exit. Continue waiting for running or scheduled tasks. Cover empty/completed views and negative controls with the offline PTY regression and update the testing coverage docs. --- tests/README.md | 4 ++- tests/integration/README.md | 9 ++++-- tests/integration/utils/terminal.py | 32 +++++++++++++++++++-- tests/test_e2e_cuj_helpers.py | 44 +++++++++++++++++++++++------ 4 files changed, 76 insertions(+), 13 deletions(-) diff --git a/tests/README.md b/tests/README.md index 23e32a691..9d6ed82d0 100644 --- a/tests/README.md +++ b/tests/README.md @@ -21,7 +21,9 @@ The live fixture compares configuration before and after the journey, even on fa CUJs never republish configuration or create a remote reservation. CUJ helper tests also verify that unsupported agent names fail rather than defaulting to Codex. Offline PTY checks verify that the Claude background-task wait observes "No tasks currently -running" before sending `/exit`, without stopping tasks or confirming an exit dialog. +running" or a native task menu containing only completed rows before sending `/exit`, without +stopping tasks or confirming an exit dialog. Running/scheduled sections, incomplete row counts, +and completed-task text outside the native menu cannot satisfy the wait. The live Claude subagent skill-toggle journey uses this wait after its final calculation. They cover Claude/Codex helper dispatch and rejection of routing decisions without the agent-specific prompt-submission evidence. diff --git a/tests/integration/README.md b/tests/integration/README.md index 6bfd036bb..89651d81d 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -47,9 +47,14 @@ The existing unit tests keep their fixtures. Integration has an independent pytest configuration and uses `--confcutdir` so those fixtures cannot leak in. It is not collected by the default `uv run pytest` command. The Claude subagent skill-toggle journey opens `/tasks` after its final calculation -and waits up to 180 seconds for "No tasks currently running" before submitting `/exit`. +and waits up to 180 seconds for "No tasks currently running" or a native menu containing only +completed task rows before submitting `/exit`. Claude 2.1.290 retains finished children in a +`Completed (N)` list, so the empty-state message is not required when the list has exactly N +checkmarked `done` rows and no running or scheduled section. The helper observes menu dismissal +after Escape before typing `/exit`. Offline PTY checks reject active/scheduled tasks, partial +lists, and completed-task text outside the native menu. It neither stops tasks nor confirms an exit dialog. Process exit retains its 30-second -timeout. Offline PTY checks cover this ordering; a live 2.1.290 run is still required. +timeout. Offline PTY checks cover this ordering. `TestChildStdoutLaunch` in `../test_cli.py` covers clean Claude print-mode and Codex exec/app-server stdout, early launch errors, and forwarding through ug's `--`. diff --git a/tests/integration/utils/terminal.py b/tests/integration/utils/terminal.py index 11b412ff9..299561298 100644 --- a/tests/integration/utils/terminal.py +++ b/tests/integration/utils/terminal.py @@ -19,6 +19,31 @@ from .evidence import agent_sessions, assert_no_terminal_api_error +def _claude_background_task_menu(text): + return re.search( + r"(?ms)^[ \t]*Background[ \t]*\n(.*?)" + r"^[ \t]*↑/↓ to select · Enter to view · Esc to close[ \t]*\s*\Z", + text, + ) + + +def _claude_background_tasks_complete(text): + if re.search(r"(?m)^\s*No tasks currently running\s*$", text): + return True + menu = _claude_background_task_menu(text) + if menu is None: + return False + rows = [line.strip() for line in menu[1].splitlines() if line.strip()] + if not rows: + return False + completed = re.fullmatch(r"Completed \(([1-9][0-9]*)\)", rows[0]) + return ( + completed is not None + and len(rows) - 1 == int(completed[1]) + and all(re.fullmatch(r"(?:❯\s*)?✔\s+.+\s+done\s+·\s+.+", row) for row in rows[1:]) + ) + + class TerminalScreen(pyte.Screen): def __init__(self, columns, lines, send): super().__init__(columns, lines) @@ -405,13 +430,16 @@ def wait_for_background_tasks(self, timeout=180): assert self.agent == "claude", self.agent self.submit("/tasks") self.wait_for( - lambda text: re.search(r"(?m)^\s*No tasks currently running\s*$", text), + _claude_background_tasks_complete, "Claude's task view reporting no running tasks", timeout=timeout, ) self.send("\x1b", "close the completed background-task view") self.wait_for( - lambda text: "No tasks currently running" not in text, + lambda text: ( + "No tasks currently running" not in text + and _claude_background_task_menu(text) is None + ), "the prompt after closing the background-task view", ) diff --git a/tests/test_e2e_cuj_helpers.py b/tests/test_e2e_cuj_helpers.py index 62892fcfd..5f7e219ca 100644 --- a/tests/test_e2e_cuj_helpers.py +++ b/tests/test_e2e_cuj_helpers.py @@ -211,9 +211,40 @@ def test_wait_until_acknowledges_ready_billing_notices_and_observes_dismissal(tm @requires_pty -@pytest.mark.parametrize("complete", [True, False]) -def test_claude_waits_for_background_task_completion_before_exit(tmp_path, complete): +@pytest.mark.parametrize( + "view", ["empty", "completed", "running", "mixed", "scheduled", "partial", "transcript"] +) +def test_claude_waits_for_background_task_completion_before_exit(tmp_path, view): """An offline terminal fixture requires /tasks, completion, Escape, then /exit.""" + views = { + "empty": "No tasks currently running", + "completed": ( + "Background\n\n Completed (2)\n" + "❯ ✔ ug_subagent_two done · Haiku 4.5\n" + " ✔ ug_subagent_one done · Sonnet 5\n\n" + "↑/↓ to select · Enter to view · Esc to close" + ), + "running": ( + "Background\n\n Running (1)\n❯ ug_subagent_one running · Sonnet 5\n\n" + "↑/↓ to select · Enter to view · Esc to close" + ), + "mixed": ( + "Background\n\n Running (1)\n❯ ug_subagent_one running · Sonnet 5\n" + " Completed (1)\n ✔ ug_subagent_two done · Haiku 4.5\n\n" + "↑/↓ to select · Enter to view · Esc to close" + ), + "scheduled": ( + "Background\n\n Scheduled (1)\n❯ scheduled task · Runs once in 1m\n" + " Completed (1)\n ✔ ug_subagent_two done · Haiku 4.5\n\n" + "↑/↓ to select · Enter to view · Esc to close" + ), + "partial": ( + "Background\n\n Completed (2)\n❯ ✔ ug_subagent_one done · Sonnet 5\n\n" + "↑/↓ to select · Enter to view · Esc to close" + ), + "transcript": ("Background\n\n Completed (1)\n✔ ug_subagent_one done · Sonnet 5\n\n❯"), + } + complete = view in ("empty", "completed") script = tmp_path / "ug" script.write_text( f"""#!{sys.executable} @@ -225,13 +256,10 @@ def test_claude_waits_for_background_task_completion_before_exit(tmp_path, compl print("❯", flush=True) assert sys.stdin.readline().strip() == "/tasks" print("Background tasks: scheduled task · Runs once in 1m", flush=True) -if not {complete!r}: - time.sleep(30) - raise SystemExit(1) time.sleep(0.8) previous = termios.tcgetattr(sys.stdin.fileno()) tty.setraw(sys.stdin.fileno()) -print("\\x1b[2J\\x1b[HNo tasks currently running", flush=True) +print("\\x1b[2J\\x1b[H" + {views[view]!r}.replace("\\n", "\\r\\n"), flush=True) assert sys.stdin.read(1) == "\\x1b" termios.tcsetattr(sys.stdin.fileno(), termios.TCSANOW, previous) print("\\x1b[2J\\x1b[H❯", flush=True) @@ -250,10 +278,10 @@ def test_claude_waits_for_background_task_completion_before_exit(tmp_path, compl for action in tui.actions if action["reason"] == "close the completed background-task view" ) - assert "No tasks currently running" in close["screen_before"] + assert views[view].splitlines()[0] in close["screen_before"] else: with pytest.raises(AssertionError, match="task view reporting no running tasks"): - tui.wait_for_background_tasks(timeout=0.3) + tui.wait_for_background_tasks(timeout=2) assert not tui.ended assert not any("/exit" in action["keys"] for action in tui.actions) From 09548c6a017e94fbd1794d90b7ff7de64be440a8 Mon Sep 17 00:00:00 2001 From: "lilly.luo" Date: Fri, 9 Oct 2026 17:59:40 +0000 Subject: [PATCH 28/30] Wait for Claude background work before CUJ4 exit --- tests/README.md | 2 ++ tests/e2e_cuj/README.md | 3 +++ tests/e2e_cuj/test_cuj4_smart_routing.py | 2 ++ tests/integration/README.md | 2 ++ tests/test_e2e_cuj_helpers.py | 14 ++++++++++++-- 5 files changed, 21 insertions(+), 2 deletions(-) diff --git a/tests/README.md b/tests/README.md index 9d6ed82d0..20babbb86 100644 --- a/tests/README.md +++ b/tests/README.md @@ -25,6 +25,8 @@ running" or a native task menu containing only completed rows before sending `/e stopping tasks or confirming an exit dialog. Running/scheduled sections, incomplete row counts, and completed-task text outside the native menu cannot satisfy the wait. The live Claude subagent skill-toggle journey uses this wait after its final calculation. +CUJ4's Claude preset sessions also use it after verifying delegated file tasks: a completed +answer does not prove that a resumed child has stopped running. They cover Claude/Codex helper dispatch and rejection of routing decisions without the agent-specific prompt-submission evidence. CUJ3 and CUJ4 also verify Claude's native recovery from the known thinking-display 400: diff --git a/tests/e2e_cuj/README.md b/tests/e2e_cuj/README.md index 90fd648a6..0d0b24f2b 100644 --- a/tests/e2e_cuj/README.md +++ b/tests/e2e_cuj/README.md @@ -103,6 +103,9 @@ before continuing. Later occurrences are handled the same way; any other dialog Claude's native peer hand-back remains part of the parent's turn only when its sender matches a successful Agent spawn in that turn. The hand-back alone cannot satisfy completion: the parent must still produce its own final answer with the hidden value. +After verifying the delegated task, Claude sessions also wait in the native `/tasks` view +until no background work remains, including a child resumed after its initial answer. +Only then does the test close the task view and submit `/exit`. Each preset also runs public `ug revert` in cleanup, including after a failed assertion, so interactive launches' OS-managed settings cannot contaminate the next preset's configuration. The existing Claude explicit-model precedence case remains skipped; the routing-disabled case diff --git a/tests/e2e_cuj/test_cuj4_smart_routing.py b/tests/e2e_cuj/test_cuj4_smart_routing.py index 235cf599e..27b9f8588 100644 --- a/tests/e2e_cuj/test_cuj4_smart_routing.py +++ b/tests/e2e_cuj/test_cuj4_smart_routing.py @@ -344,6 +344,8 @@ def test_smart_router_config_version( assert canonical_model(inference.payload["model"]) == canonical_model( selections[0]["route_option"]["model"] ) + if agent == CLAUDE: + tui.wait_for_background_tasks() tui.exit_normally() finally: if previous is None: diff --git a/tests/integration/README.md b/tests/integration/README.md index 89651d81d..52dea8683 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -55,6 +55,8 @@ after Escape before typing `/exit`. Offline PTY checks reject active/scheduled t lists, and completed-task text outside the native menu. It neither stops tasks nor confirms an exit dialog. Process exit retains its 30-second timeout. Offline PTY checks cover this ordering. +The shared wait also runs before CUJ4's Claude preset sessions exit, after delegated-task +evidence is checked, because Claude can resume a child after its first completed answer. `TestChildStdoutLaunch` in `../test_cli.py` covers clean Claude print-mode and Codex exec/app-server stdout, early launch errors, and forwarding through ug's `--`. diff --git a/tests/test_e2e_cuj_helpers.py b/tests/test_e2e_cuj_helpers.py index 5f7e219ca..f1109cc19 100644 --- a/tests/test_e2e_cuj_helpers.py +++ b/tests/test_e2e_cuj_helpers.py @@ -212,7 +212,8 @@ def test_wait_until_acknowledges_ready_billing_notices_and_observes_dismissal(tm @requires_pty @pytest.mark.parametrize( - "view", ["empty", "completed", "running", "mixed", "scheduled", "partial", "transcript"] + "view", + ["empty", "completed", "resumed", "running", "mixed", "scheduled", "partial", "transcript"], ) def test_claude_waits_for_background_task_completion_before_exit(tmp_path, view): """An offline terminal fixture requires /tasks, completion, Escape, then /exit.""" @@ -244,11 +245,13 @@ def test_claude_waits_for_background_task_completion_before_exit(tmp_path, view) ), "transcript": ("Background\n\n Completed (1)\n✔ ug_subagent_one done · Sonnet 5\n\n❯"), } - complete = view in ("empty", "completed") + views["resumed"] = views["mixed"] + complete = view in ("empty", "completed", "resumed") script = tmp_path / "ug" script.write_text( f"""#!{sys.executable} import sys +import select import termios import time import tty @@ -260,6 +263,10 @@ def test_claude_waits_for_background_task_completion_before_exit(tmp_path, view) previous = termios.tcgetattr(sys.stdin.fileno()) tty.setraw(sys.stdin.fileno()) print("\\x1b[2J\\x1b[H" + {views[view]!r}.replace("\\n", "\\r\\n"), flush=True) +if {view == "resumed"!r}: + time.sleep(0.8) + assert not select.select([sys.stdin], [], [], 0)[0], "Exited with a resumed child running" + print("\\x1b[2J\\x1b[H" + {views["completed"]!r}.replace("\\n", "\\r\\n"), flush=True) assert sys.stdin.read(1) == "\\x1b" termios.tcsetattr(sys.stdin.fileno(), termios.TCSANOW, previous) print("\\x1b[2J\\x1b[H❯", flush=True) @@ -279,6 +286,9 @@ def test_claude_waits_for_background_task_completion_before_exit(tmp_path, view) if action["reason"] == "close the completed background-task view" ) assert views[view].splitlines()[0] in close["screen_before"] + if view == "resumed": + assert "Running (1)" not in close["screen_before"] + assert "Completed (2)" in close["screen_before"] else: with pytest.raises(AssertionError, match="task view reporting no running tasks"): tui.wait_for_background_tasks(timeout=2) From 995eca1ff05dad4a1332cdb25453b3123fb72aa0 Mon Sep 17 00:00:00 2001 From: "lilly.luo" Date: Fri, 9 Oct 2026 18:14:54 +0000 Subject: [PATCH 29/30] Wait for routed child evidence without reconstructing parent turns --- tests/README.md | 11 +-- tests/e2e_cuj/README.md | 10 +- tests/e2e_cuj/helpers/evidence.py | 41 +------- tests/e2e_cuj/test_cuj4_smart_routing.py | 7 +- tests/integration/README.md | 6 +- tests/test_cuj_evidence.py | 117 ----------------------- tests/test_integration_evidence.py | 14 +++ 7 files changed, 28 insertions(+), 178 deletions(-) diff --git a/tests/README.md b/tests/README.md index 20babbb86..0e587d36d 100644 --- a/tests/README.md +++ b/tests/README.md @@ -33,13 +33,10 @@ CUJ3 and CUJ4 also verify Claude's native recovery from the known thinking-displ the same payload without display must receive a non-empty 200 for adaptive or enabled thinking. If the next attempt rejects `safeguards`, only its native removal is accepted before the final 200. Offline regressions reject missing/failed retries and changes to the model, prompt, budget, or effort. -Native evidence-reader regressions cover Claude's background-agent completion notifications; -notifications alone cannot substitute for a final parent answer. These offline checks do not -establish a live routing pass. -Claude peer hand-backs are correlated with successful Agent spawns in the same parent turn; -unknown senders, mismatched sessions, failed spawns, and reports without a final parent answer -are rejected. Offline PTY checks cover billing-notice input readiness, observed dismissal, -and a later notice for child work. +Delegated tasks wait for the native child's answer, independently of the parent's final turn. +Offline checks require the child's hidden file value; a parent-only answer cannot satisfy the wait. +First-prompt tasks retain completed parent-turn evidence. Offline PTY checks cover billing-notice +input readiness, observed dismissal, and a later notice for child work. Child HTTP matching accepts Claude's single appended transport newline after the routing checkpoint; additional text, whitespace, and Codex prompt changes remain rejected. Offline CUJ regressions cover Codex delegated-turn completion and exact task-request matching; diff --git a/tests/e2e_cuj/README.md b/tests/e2e_cuj/README.md index 0d0b24f2b..5c7d34189 100644 --- a/tests/e2e_cuj/README.md +++ b/tests/e2e_cuj/README.md @@ -89,10 +89,11 @@ Parent and child requests use the same verified thinking-display recovery as CUJ only known display/safeguards rejections followed by native removal of the rejected field and a final non-empty 200 are accepted. Model, prompt, budget, and effort must remain unchanged. Each preset session then explicitly requests one child for a separate hidden-value -file task. Assertions require a native child transcript containing the value, the completed -parent answer, a correlated spawn-routing decision, and successful child inference on the +file task. Assertions require a new native child transcript containing the value, +a correlated spawn-routing decision, and successful child inference on the router's selected model. This tests requested delegation, not automatic orchestrator delegation. -Codex requires a matching completed turn and final parent answer, not just child notifications. +The wait uses the child's answer directly; parent-answer reconstruction is reserved for first-prompt tasks. +Offline checks reject parent-only answers and accept child answers before the parent replies. Offline evidence regressions do not establish a live CUJ pass. Selector cases have separate TUI artifact names, and the session environment is restored afterward. Claude sessions start in auto mode; unless a routed first prompt switches the session to Haiku @@ -100,9 +101,6 @@ Claude sessions start in auto mode; unless a routed first prompt switches the se Claude's informational auto-mode classifier billing notice over the transcript. The CUJ terminal waits for that exact notice to render stably, presses Enter (continue), and observes dismissal before continuing. Later occurrences are handled the same way; any other dialog still fails. -Claude's native peer hand-back remains part of the parent's turn only when its sender matches -a successful Agent spawn in that turn. The hand-back alone cannot satisfy completion: the -parent must still produce its own final answer with the hidden value. After verifying the delegated task, Claude sessions also wait in the native `/tasks` view until no background work remains, including a child resumed after its initial answer. Only then does the test close the task view and submit `/exit`. diff --git a/tests/e2e_cuj/helpers/evidence.py b/tests/e2e_cuj/helpers/evidence.py index 4371918f4..1cde4346d 100644 --- a/tests/e2e_cuj/helpers/evidence.py +++ b/tests/e2e_cuj/helpers/evidence.py @@ -188,52 +188,13 @@ def _completed_turn(records, task): first = records[prompts[0]] session_id = first.get("sessionId") responses = [] - spawn_tools = set() - spawned_children = set() for row in records[prompts[0] + 1 :]: msg = row.get("message", {}) - origin = row.get("origin", {}) - child_report = ( - origin.get("kind") == "peer" - and origin.get("handback") is True - and origin.get("from") in spawned_children - and origin.get("senderTaskId") == origin.get("from") - and row.get("sessionId") == session_id - ) - if ( - row.get("type") == "user" - and isinstance(msg.get("content"), str) - and origin.get("kind") != "task-notification" - and not child_report - ): + if row.get("type") == "user" and isinstance(msg.get("content"), str): break - if row.get("type") == "user" and isinstance(msg.get("content"), list): - result = row.get("toolUseResult", {}) - if ( - row.get("sessionId") == session_id - and result.get("status") == "async_launched" - and result.get("isAsync") is True - and isinstance(result.get("agentId"), str) - and result["agentId"] - and any( - part.get("type") == "tool_result" - and part.get("tool_use_id") in spawn_tools - and not part.get("is_error") - for part in msg["content"] - ) - ): - spawned_children.add(result["agentId"]) if row.get("type") == "assistant": assert row.get("sessionId") == session_id and session_id responses.append(msg) - spawn_tools.update( - part["id"] - for part in msg.get("content", []) - if part.get("type") == "tool_use" - and part.get("name") == "Agent" - and isinstance(part.get("id"), str) - and part["id"] - ) if not responses: return None last = responses[-1] diff --git a/tests/e2e_cuj/test_cuj4_smart_routing.py b/tests/e2e_cuj/test_cuj4_smart_routing.py index 27b9f8588..c19ab368e 100644 --- a/tests/e2e_cuj/test_cuj4_smart_routing.py +++ b/tests/e2e_cuj/test_cuj4_smart_routing.py @@ -298,14 +298,13 @@ def test_smart_router_config_version( ) decision_count = len(read_jsonl(decisions_path)) checkpoint = recorder.checkpoint() + existing_sessions = set(agent_sessions(session, agent)) tui.submit(child_task.prompt) - tui.task(evidence, child_task) - tui.wait_for( - lambda screen, task=child_task, agent=agent: task.completed( + tui.wait_until( + lambda task=child_task, agent=agent: task.completed( session, agent, child=True ), "completed native subagent file task", - timeout=240, ) children = { path: records diff --git a/tests/integration/README.md b/tests/integration/README.md index 52dea8683..fb3b520e6 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -22,10 +22,8 @@ The native retry may also encounter `safeguards: Extra inputs are not permitted` requires the next recorded attempt to remove only `safeguards` and reach a non-empty 200. Offline regressions check both CUJs and reject changes to the model, prompt, budget, or effort. Each preset cleans up interactive OS-managed settings with public `ug revert`, even on failure. -Offline transcript tests check native Claude background-agent completion evidence; -a completion notification alone is not treated as the parent's completed answer. -Native peer hand-backs also remain in the parent turn when correlated with a successful -Agent spawn; the parent must still answer. Offline negative controls reject unrelated reports. +Delegated CUJ tasks wait for the native child's hidden file value, independently of the parent's +final turn. Offline checks reject parent-only answers. First-prompt tasks retain parent-turn evidence. The CUJ terminal waits for stable billing-notice rendering before Enter and observes dismissal; an offline PTY regression also exercises a later notice during child work. Claude child HTTP prompts may carry one appended newline after the route-selection checkpoint; diff --git a/tests/test_cuj_evidence.py b/tests/test_cuj_evidence.py index a60d22761..40f84ba74 100644 --- a/tests/test_cuj_evidence.py +++ b/tests/test_cuj_evidence.py @@ -103,66 +103,6 @@ def test_cuj_evidence_completed_native_turn(tmp_path, agent): assert result["selected_model"] == model -def test_cuj_evidence_claude_async_notification_keeps_parent_turn(): - task = SimpleNamespace( - prompt="Delegate reading the file to one subagent.", value="hidden-value" - ) - prompt, answer = records(CLAUDE, task, "system.ai.claude-haiku-4-5") - delegation = copy.deepcopy(answer) - delegation["message"].update( - id="delegation", - stop_reason="tool_use", - content=[ - {"type": "tool_use", "id": "tool", "name": "Agent", "input": {"prompt": task.prompt}} - ], - ) - waiting = copy.deepcopy(answer) - waiting["message"].update( - id="waiting", content=[{"type": "text", "text": "Waiting for the child."}] - ) - rows = [ - prompt, - delegation, - { - "type": "user", - "sessionId": "session", - "message": { - "content": [{"type": "tool_result", "tool_use_id": "tool", "content": task.value}] - }, - }, - waiting, - { - "type": "user", - "sessionId": "session", - "origin": {"kind": "task-notification"}, - "promptSource": "system", - "turnOrigin": "task_notification", - "message": { - "content": ( - "\ncompleted\n" - f"{task.value}\n" - ) - }, - }, - answer, - ] - assert completed_turn(CLAUDE, rows[:3], task) is None - assert completed_turn(CLAUDE, rows[:-1], task) is None - turn = completed_turn(CLAUDE, rows, task) - assert turn is not None - assert (turn.session_id, turn.turn_id, turn.answer) == ("session", "response", task.value) - assert turn.models == ["system.ai.claude-haiku-4-5"] * 3 - child = copy.deepcopy(rows) - child[-1]["isSidechain"] = True - assert completed_turn(CLAUDE, child, task) is None - with pytest.raises(AssertionError, match="Prompt was submitted more than once"): - completed_turn(CLAUDE, [*rows, prompt], task) - missing_model = copy.deepcopy(rows) - del missing_model[3]["message"]["model"] - with pytest.raises(AssertionError, match="Missing inference model metadata"): - completed_turn(CLAUDE, missing_model, task) - - def test_cuj_evidence_claude_real_user_message_ends_parent_turn(): task = SimpleNamespace( prompt="Delegate reading the file to one subagent.", value="hidden-value" @@ -181,63 +121,6 @@ def test_cuj_evidence_claude_real_user_message_ends_parent_turn(): assert completed_turn(CLAUDE, [prompt, user, answer], task) is None -@pytest.mark.parametrize( - "failure", - [ - None, - "no_answer", - "unknown_child", - "wrong_sender", - "wrong_session", - "not_handback", - "failed_spawn", - "unrelated_tool", - ], -) -def test_cuj_evidence_claude_spawned_child_handback_keeps_parent_turn(failure): - task = SimpleNamespace(prompt="Delegate reading the file.", value="hidden-value") - prompt, answer = records(CLAUDE, task, "system.ai.claude-sonnet-4-6") - delegation = copy.deepcopy(answer) - delegation["message"].update( - id="delegation", - stop_reason="tool_use", - content=[{"type": "tool_use", "id": "spawn", "name": "Agent", "input": {}}], - ) - spawned = { - "type": "user", - "sessionId": "session", - "message": {"content": [{"type": "tool_result", "tool_use_id": "spawn"}]}, - "toolUseResult": {"status": "async_launched", "isAsync": True, "agentId": "child"}, - } - handback = { - "type": "user", - "sessionId": "session", - "origin": {"kind": "peer", "handback": True, "from": "child", "senderTaskId": "child"}, - "message": {"content": f"Another Claude session sent a message: {task.value}"}, - } - if failure == "unknown_child": - handback["origin"].update({"from": "other", "senderTaskId": "other"}) - elif failure == "wrong_sender": - handback["origin"]["senderTaskId"] = "other" - elif failure == "wrong_session": - handback["sessionId"] = "other" - elif failure == "not_handback": - handback["origin"]["handback"] = False - elif failure == "failed_spawn": - spawned["message"]["content"][0]["is_error"] = True - elif failure == "unrelated_tool": - delegation["message"]["content"][0]["name"] = "Read" - rows = [prompt, delegation, spawned, handback] - if failure != "no_answer": - rows.append(answer) - turn = completed_turn(CLAUDE, rows, task) - if failure is None: - assert turn is not None - assert (turn.turn_id, turn.answer) == ("response", task.value) - else: - assert turn is None - - @pytest.mark.parametrize("agent", [CLAUDE, CODEX]) @pytest.mark.parametrize( "failure", diff --git a/tests/test_integration_evidence.py b/tests/test_integration_evidence.py index cf04efc76..e0c22846d 100644 --- a/tests/test_integration_evidence.py +++ b/tests/test_integration_evidence.py @@ -7,6 +7,7 @@ from tests.integration.utils import evidence from tests.integration.utils.agents import claude, codex from tests.integration.utils.evidence import ( + FileTask, SubagentCalculation, assert_no_terminal_api_error, assistant_answer_contains, @@ -103,6 +104,19 @@ def test_tagged_calculation_requires_the_native_child_answer(tmp_path, agent): assert "1+1" in task.prompt +@pytest.mark.parametrize("agent", ["claude", "codex"]) +@pytest.mark.parametrize("child", [False, True]) +def test_file_task_child_completion_is_independent_of_parent_answer(tmp_path, agent, child): + session = _Session(tmp_path) + session.cwd = tmp_path + task = FileTask(session) + assert task.value not in task.delegate_prompt + assert not task.completed(session, agent, child=True) + _write_answer(tmp_path, agent, child=child, value=task.value) + assert task.completed(session, agent, child=True) == child + assert task.completed(session, agent) != child + + def test_codex_model_identity_uses_only_the_completed_answer_turn(): records = [ { From 2a44f77bf5ad82bd71aaa78938f6b9552a47aedb Mon Sep 17 00:00:00 2001 From: "lilly.luo" Date: Fri, 9 Oct 2026 18:26:59 +0000 Subject: [PATCH 30/30] Correlate Claude compatibility retries with their agent context --- tests/README.md | 1 + tests/e2e_cuj/README.md | 1 + tests/e2e_cuj/helpers/evidence.py | 6 +++++- tests/integration/README.md | 3 ++- tests/test_e2e_cuj_helpers.py | 23 +++++++++++++++++++++++ 5 files changed, 32 insertions(+), 2 deletions(-) diff --git a/tests/README.md b/tests/README.md index 0e587d36d..d3b8db57d 100644 --- a/tests/README.md +++ b/tests/README.md @@ -32,6 +32,7 @@ the agent-specific prompt-submission evidence. CUJ3 and CUJ4 also verify Claude's native recovery from the known thinking-display 400: the same payload without display must receive a non-empty 200 for adaptive or enabled thinking. If the next attempt rejects `safeguards`, only its native removal is accepted before the final 200. +Retries match system context and session metadata so concurrent parent traffic is excluded. Offline regressions reject missing/failed retries and changes to the model, prompt, budget, or effort. Delegated tasks wait for the native child's answer, independently of the parent's final turn. Offline checks require the child's hidden file value; a parent-only answer cannot satisfy the wait. diff --git a/tests/e2e_cuj/README.md b/tests/e2e_cuj/README.md index 5c7d34189..3382630a2 100644 --- a/tests/e2e_cuj/README.md +++ b/tests/e2e_cuj/README.md @@ -37,6 +37,7 @@ The shared recovery check covers both adaptive and enabled thinking, preserving Claude may then receive `safeguards: Extra inputs are not permitted`. Recovery must remove only that field in the next native attempt and end with a non-empty HTTP 200. The checks inspect recorded traffic without modifying or replaying requests. +Retries match system context and session metadata to exclude concurrent parent traffic. The test class selects the CUJ3 workspace, `https://dbc-bbdd5508-648e.cloud.databricks.com`. The shared `cuj` fixture supplies its authenticated SDK client and isolated local session; diff --git a/tests/e2e_cuj/helpers/evidence.py b/tests/e2e_cuj/helpers/evidence.py index 1cde4346d..73a401474 100644 --- a/tests/e2e_cuj/helpers/evidence.py +++ b/tests/e2e_cuj/helpers/evidence.py @@ -92,11 +92,15 @@ def served_inference_request(recorder, requests, request, agent): assert_served(recorder, request, model) return request following = requests[requests.index(request) + 1 :] + # Parent and child traffic can interleave on the same endpoint. retry = next( ( candidate for candidate in following - if candidate.method == request.method and candidate.path == request.path + if candidate.method == request.method + and candidate.path == request.path + and candidate.payload.get("system") == payload.get("system") + and candidate.payload.get("metadata") == payload.get("metadata") ), None, ) diff --git a/tests/integration/README.md b/tests/integration/README.md index fb3b520e6..8f9171276 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -16,7 +16,8 @@ managed-default and explicit-model cases unchanged. One additional test runs the context in inference input, and completed explicitly requested routed subagents. This is not automatic orchestrator-delegation coverage. No workspace configuration is modified. CUJ3 and CUJ4 require a non-empty HTTP 200 for inference. Claude 2.1.290's known -thinking-display 400 is accepted only when the next inference request removes `display`, +thinking-display 400 is accepted only when the next inference request with the same system +context and session metadata removes `display`, preserves every other payload field, and succeeds; adaptive and enabled thinking are covered. The native retry may also encounter `safeguards: Extra inputs are not permitted`; that requires the next recorded attempt to remove only `safeguards` and reach a non-empty 200. diff --git a/tests/test_e2e_cuj_helpers.py b/tests/test_e2e_cuj_helpers.py index f1109cc19..cf32874a1 100644 --- a/tests/test_e2e_cuj_helpers.py +++ b/tests/test_e2e_cuj_helpers.py @@ -514,6 +514,29 @@ def test_served_inference_request_keeps_successful_first_attempt(thinking_displa assert served_inference_request(recorder, requests, requests[0], agent) is requests[0] +@pytest.mark.parametrize("context", ["system", "metadata"]) +@pytest.mark.parametrize("changed_model", [False, True]) +def test_served_inference_matches_retry_despite_interleaved_traffic( + thinking_display_exchange, context, changed_model +): + recorder, requests, responses, _, _ = thinking_display_exchange + rejected, retry = requests + unrelated = SimpleNamespace( + sequence=2, + method=rejected.method, + path=rejected.path, + payload={**retry.payload, context: "parent context"}, + ) + requests.insert(1, unrelated) + responses.insert(1, SimpleNamespace(status_code=200, headers={}, body=b"parent stream")) + if changed_model: + retry.payload["model"] = "wrong model" + with pytest.raises(AssertionError, match="changed more than the rejected field"): + served_inference_request(recorder, requests, rejected, CLAUDE) + else: + assert served_inference_request(recorder, requests, rejected, CLAUDE) is retry + + @pytest.fixture def safeguards_exchange(thinking_display_exchange): recorder, requests, responses, task, model = thinking_display_exchange