diff --git a/AGENTS.md b/AGENTS.md index 156609e5f..ed8b06ac2 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -29,6 +29,56 @@ Tests live in `tests/`. - Add or update focused tests for behavior changes. - Do not modify generated or lock files unless the dependency graph intentionally changes. +## Smart-routing configuration + +`SMART_ROUTER_CONFIG_VERSION` is the external selector. Version definitions live in +`src/ucode/smart_routing/config.py` under `_VERSIONS`; their managed environment keys +are registered in `SMART_ROUTING_ENV_KEYS` in `src/ucode/constants.py`. +An unset or empty selector preserves legacy environment-flag behavior. +A valid nonempty selector overrides every conflicting legacy value in that registry. +Do not use `setdefault` or preserve inherited values for version-owned parameters. +Resolve and materialize valid selectors at the CLI boundary before argument parsing or callbacks. +Unknown selectors are a no-op: do not apply a preset, mutate the inherited environment, or raise +an error. Existing legacy flags and explicit launch/session controls retain their behavior. +Restore the inherited environment on every exit. +Explicit launch/session on/off controls still apply after version expansion. +Managed routing defaults must not rewrite already-resolved version flags. + +### Adding a parameter + +1. Define its environment-variable constant in `src/ucode/constants.py` and add it to + `SMART_ROUTING_ENV_KEYS`. +2. Set an explicit value for it in **every** `_VERSIONS` entry, including existing versions. + Choose values that preserve existing versions' behavior. Current parameters accept only + the strings `"0"` and `"1"`; do not use booleans, empty strings, or omitted keys. +3. Add its consumer in the appropriate routing module. Use `resolve_environment` for + config-aware reads, or the legacy flags materialized by `apply_config` at launch. + Keep launch-scoped changes restorable and preserve legacy behavior without a selector. +4. `SMART_ROUTING_ENV_KEYS` controls environment snapshots, restoration, and launch/session + off overrides. Keep routing activation limited to the V2 and subagent-only flags: + orchestration alone must not enable routing. Add regression tests for the new parameter's + controls. Session overrides must apply after version resolution. + +`_validate_versions` runs at module import and rejects missing keys, unknown keys, and +invalid values. Do not weaken the complete-key check or infer required keys from `_VERSIONS`. +If a new parameter needs nonbinary values, add parameter-specific validation and tests. + +### Adding a config type or revision + +1. Add a complete mapping to `_VERSIONS` with an explicit suffix, such as `new_mode_v0`. + For a changed existing mode, add `existing_mode_v1` rather than changing its `_v0` behavior. + Do not add unsuffixed names or version aliases. +2. Define every key in `SMART_ROUTING_ENV_KEYS`; never rely on the caller's + inherited environment to fill missing values. +3. Update the version table and examples in `README.md` and any affected bundled-skill docs. +4. Extend `tests/test_smart_routing_config.py` for the new mode, precedence over legacy flags, + environment restoration, validation failures, and applicable session/launch behavior. + Update unknown-version tests when a previously rejected version becomes supported. + Follow `tests/AGENTS.md` and update its coverage READMEs. + +Run `uv run pytest tests/test_smart_routing_config.py` plus relevant routing/CLI tests, +then `just lint`. These component checks do not establish live agent or gateway coverage. + ## Style - Keep user-facing CLI errors actionable. diff --git a/README.md b/README.md index 7bde76c7b..59374ecdf 100644 --- a/README.md +++ b/README.md @@ -251,15 +251,55 @@ The generated shell hooks expect Git Bash; PowerShell-only setups are not covere ### Smart Router Orchestrator -Smart-routed Claude and Codex sessions install `smart-router`. Set -`ENABLE_SMART_ROUTER_ORCHESTRATOR=1` at launch to also install and activate Smart Router -Orchestrator through the bundled `smart-router-orchestrator` skill; orchestration is off by -default. For example: +Use `SMART_ROUTER_CONFIG_VERSION` at launch to select a smart-routing configuration: + +| Version | Subagent routing | First-prompt routing | Orchestrator | +| --- | --- | --- | --- | +| `first_prompt_and_subagent_no_orch_v0` | On | On | Off | +| `subagent_only_v0` | On | Off | Off | +| `subagent_only_v1` | On | Off | Off | +| `subagent_orch_v0` | On | Off | On | +| `subagent_orch_v1` | On | Off | On | + +`first_prompt_and_subagent_no_orch_v0` is the customer configuration for first-prompt +and subagent routing without orchestration: `ENABLE_SMART_ROUTING_V2=1`, +`ENABLE_SMART_ROUTING_SUBAGENT_ONLY=0`, and `ENABLE_SMART_ROUTER_ORCHESTRATOR=0`. + +`subagent_only_v1` sets both `ENABLE_SMART_ROUTING_V2` and +`ENABLE_SMART_ROUTING_SUBAGENT_ONLY` to `"1"`. Subagent-only takes precedence, +so first-prompt routing remains off; orchestration is also off. + +`subagent_orch_v1` enables all three legacy flags. Like `subagent_only_v1`, it routes +subagents rather than the first prompt, and it additionally enables orchestration. + +`SMART_ROUTER_NAME` still selects the router independently of the preset. + +Smart-routed Claude and Codex sessions install `smart-router`. The `subagent_orch_v0` +and `subagent_orch_v1` versions also install and activate the bundled `smart-router-orchestrator` skill. +For example: ```bash -ENABLE_SMART_ROUTER_ORCHESTRATOR=1 ENABLE_SMART_ROUTING_SUBAGENT_ONLY=1 ug claude +SMART_ROUTER_CONFIG_VERSION=subagent_orch_v0 ug claude ``` +The version takes precedence over conflicting legacy flags. Before parsing command options +or running any command callbacks, UG expands it into +`ENABLE_SMART_ROUTING_V2`, `ENABLE_SMART_ROUTING_SUBAGENT_ONLY`, and +`ENABLE_SMART_ROUTER_ORCHESTRATOR` for the launched session. When the version is +unset or empty, these legacy flags retain their existing behavior, including +first-prompt routing through `ENABLE_SMART_ROUTING_V2=1`. Unknown versions are ignored: +no preset is applied, the inherited environment is unchanged, and commands continue normally. +Explicit launch/session on/off controls apply after expansion. Orchestration remains off by default. +Workspace smart-routing defaults do not rewrite the selected version's flags. + +Version names require an explicit suffix. Future revisions use new `_v1`, `_v2`, +etc. names without changing existing versions. + +Version definitions fail validation at module import if any flag in +`SMART_ROUTING_ENV_KEYS` is missing, has a value other than `"0"` or `"1"`, +or an unknown flag is present. Register new managed flags in that tuple and +explicitly set them in every version. + Use `ug codex` in the same command for Codex. Smart Router Orchestrator assigns bounded work to explorer, researcher, worker, tester, and reviewer roles while the root plans, integrates, and verifies results. Easy tasks and explicit requests not to delegate @@ -274,8 +314,8 @@ Once opted in, orchestration follows the existing smart-routing launch eligibili and session controls. Turning Smart Router off through its skill stops new automatic delegation; turning it on restores orchestration only in opted-in sessions. Explicit user requests for subagents still use normal harness behavior while routing is off. -Stored skill files do not activate orchestration when the feature flag is unset or -`ENABLE_SMART_ROUTER_ORCHESTRATOR=0`, or in non-routed sessions. Existing Isaac pilot gating +Stored skill files do not activate orchestration without an opted-in configuration, +or in non-routed sessions. Existing Isaac pilot gating and UG launch exclusions still apply. Hooks refresh orchestration state before each prompt and after compaction. A diff --git a/skills/smart-router-orchestrator/README.md b/skills/smart-router-orchestrator/README.md index 9e5aad4ed..c22625d0a 100644 --- a/skills/smart-router-orchestrator/README.md +++ b/skills/smart-router-orchestrator/README.md @@ -1,8 +1,13 @@ # Smart Router Orchestrator UG bundles the `smart-router-orchestrator` workflow and five Claude role definitions. -Smart-routed Claude and Codex launches install and -activate this skill alongside `smart-router` only with `ENABLE_SMART_ROUTER_ORCHESTRATOR=1`. +Smart-routed Claude and Codex launches install and activate this skill alongside +`smart-router` with `SMART_ROUTER_CONFIG_VERSION=subagent_orch_v0` or `subagent_orch_v1`. +UG expands the selected version into the session's legacy feature flags. The `_v1` revision +enables both V2 and subagent-only routing flags alongside orchestration; subagent-only still +takes precedence, so the first prompt is not routed. +The existing `ENABLE_SMART_ROUTER_ORCHESTRATOR=1` opt-in remains supported when +`SMART_ROUTER_CONFIG_VERSION` is unset; `subagent_only_v0` explicitly leaves orchestration off. The feature is off by default; routing alone installs only `smart-router`. The workflow is injected before root prompts and after compaction. The hook checks diff --git a/src/ucode/cli.py b/src/ucode/cli.py index 1f1ebfbdb..301b195e5 100644 --- a/src/ucode/cli.py +++ b/src/ucode/cli.py @@ -1387,6 +1387,22 @@ def revert() -> int: class _HelpOrderedGroup(TyperGroup): """Keep top-level help organized across commands and nested Typer apps.""" + def make_context( + self, + info_name: str | None, + args: list[str], + parent: _click.Context | None = None, + **extra: Any, + ) -> _click.Context: + previous = smart_routing_v2.apply_config() + try: + ctx = super().make_context(info_name, args, parent, **extra) + except BaseException: + smart_routing_v2.restore_smart_routing_env(previous) + raise + ctx.call_on_close(lambda: smart_routing_v2.restore_smart_routing_env(previous)) + return ctx + def list_commands(self, ctx: _click.Context) -> list[str]: commands = super().list_commands(ctx) order = {name: index for index, name in enumerate(_HELP_COMMAND_ORDER)} @@ -2478,10 +2494,11 @@ def _auto_configure_tool(tool: str, custom_oauth: CustomOAuthConfig | None = Non @contextmanager def _smart_routing_v2_flag(enabled: bool | None) -> Iterator[None]: """Apply an explicit routing choice without leaking into an embedding process.""" - if enabled is None: - yield - return - previous = smart_routing_v2.override_smart_routing(enabled) + previous = ( + smart_routing_v2.apply_config() + if enabled is None + else smart_routing_v2.override_smart_routing(enabled) + ) try: yield finally: @@ -3180,7 +3197,11 @@ def _launch_tool( ) print_success(f"Starting {TOOL_SPECS[tool]['display']}") with _smart_routing_v2_flag( - True if managed_smart_routing_enabled and smart_routing_enabled else None + True + if managed_smart_routing_enabled + and smart_routing_enabled + and not smart_routing_v2.smart_routing_enabled() + else None ): launch_agent(tool, state, ctx.args, options=launch_options) except RuntimeError as exc: @@ -3276,9 +3297,10 @@ def default( return set_dry_run(dry_run) try: - _launch_managed_default( - ctx, dry_run=dry_run, skip_preflight=skip_preflight, workspace=workspace - ) + with _smart_routing_v2_flag(None): + _launch_managed_default( + ctx, dry_run=dry_run, skip_preflight=skip_preflight, workspace=workspace + ) except typer.Exit: # `typer.Exit` subclasses RuntimeError, so it has to be re-raised ahead of the handler # below. Otherwise a launch that already reported its own error is followed by diff --git a/src/ucode/constants.py b/src/ucode/constants.py index 38fe7e774..030be1aa8 100644 --- a/src/ucode/constants.py +++ b/src/ucode/constants.py @@ -6,9 +6,11 @@ ENABLE_SMART_ROUTING_ENV_VAR = "ENABLE_SMART_ROUTING_V2" ENABLE_SUBAGENT_ROUTING_ENV_VAR = "ENABLE_SMART_ROUTING_SUBAGENT_ONLY" ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR = "ENABLE_SMART_ROUTER_ORCHESTRATOR" +SMART_ROUTER_CONFIG_VERSION_ENV_VAR = "SMART_ROUTER_CONFIG_VERSION" SMART_ROUTING_ENV_KEYS = ( ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, ) MODEL_PROVIDER_SERVICE_HEADER = "Databricks-Model-Provider-Service" diff --git a/src/ucode/smart_routing/config.py b/src/ucode/smart_routing/config.py new file mode 100644 index 000000000..2a7fe9c01 --- /dev/null +++ b/src/ucode/smart_routing/config.py @@ -0,0 +1,101 @@ +"""Resolve external smart-routing versions into backward-compatible feature flags.""" + +from __future__ import annotations + +import os +from collections.abc import Mapping, MutableMapping + +from ucode.constants import ( + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, + ENABLE_SMART_ROUTING_ENV_VAR, + ENABLE_SUBAGENT_ROUTING_ENV_VAR, + SMART_ROUTER_CONFIG_VERSION_ENV_VAR, + SMART_ROUTING_ENV_KEYS, +) + +# Customer preset: route the first prompt and subagents, without orchestration. +FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0 = "first_prompt_and_subagent_no_orch_v0" + +# Route only subagents, with V2 disabled and no orchestration. +SUBAGENT_ONLY_V0 = "subagent_only_v0" + +# Route only subagents, with both V2 and subagent-only flags enabled; no orchestration. +SUBAGENT_ONLY_V1 = "subagent_only_v1" + +# Route only subagents and inject the Smart Router Orchestrator workflow. +SUBAGENT_ORCH_V0 = "subagent_orch_v0" + +# Route only subagents with V2, subagent-only, and orchestration all enabled. +SUBAGENT_ORCH_V1 = "subagent_orch_v1" + +_VERSIONS = { + FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0: { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + }, + SUBAGENT_ONLY_V0: { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + }, + SUBAGENT_ONLY_V1: { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + }, + SUBAGENT_ORCH_V0: { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + }, + SUBAGENT_ORCH_V1: { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + }, +} + + +def _validate_versions(versions: Mapping[str, Mapping[str, str]]) -> None: + """Require every version to explicitly configure the complete managed flag set.""" + expected = set(SMART_ROUTING_ENV_KEYS) + for version, values in versions.items(): + missing = expected - values.keys() + unexpected = values.keys() - expected + if missing or unexpected: + raise ValueError( + f"Invalid smart-routing version {version!r}: " + f"missing env vars {sorted(missing)}; unexpected env vars {sorted(unexpected)}." + ) + for key, value in values.items(): + if value not in ("0", "1"): + raise ValueError( + f"Invalid smart-routing version {version!r}: " + f"{key} must be '0' or '1', got {value!r}." + ) + + +_validate_versions(_VERSIONS) + + +def resolve_environment(env: Mapping[str, str] | None = None) -> dict[str, str]: + """Expand a version before applying any launch or session-specific overrides.""" + resolved = dict(os.environ if env is None else env) + version = resolved.pop(SMART_ROUTER_CONFIG_VERSION_ENV_VAR, "").strip() + resolved.update(_VERSIONS.get(version, {})) + return resolved + + +def apply_config(env: MutableMapping[str, str] | None = None) -> dict[str, str | None]: + """Consume the launch selector, returning the values needed to restore its input.""" + target = os.environ if env is None else env + version = target.get(SMART_ROUTER_CONFIG_VERSION_ENV_VAR, "").strip() + preset = _VERSIONS.get(version) + if preset is None: + return {} + keys = (*SMART_ROUTING_ENV_KEYS, SMART_ROUTER_CONFIG_VERSION_ENV_VAR) + previous = {key: target.get(key) for key in keys} + target.update(preset) + target.pop(SMART_ROUTER_CONFIG_VERSION_ENV_VAR, None) + return previous diff --git a/src/ucode/smart_routing/orchestrator.py b/src/ucode/smart_routing/orchestrator.py index c86d4b0a5..61d9681ed 100644 --- a/src/ucode/smart_routing/orchestrator.py +++ b/src/ucode/smart_routing/orchestrator.py @@ -14,6 +14,7 @@ from ucode import skills from ucode.constants import ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR +from ucode.smart_routing.config import resolve_environment from ucode.smart_routing.hooks import sync_managed_hooks from ucode.smart_routing.session_env import effective_environment, session_env_path @@ -27,7 +28,7 @@ def feature_enabled(env: Mapping[str, str] | None = None) -> bool: - source = os.environ if env is None else env + source = resolve_environment(env) return source.get(ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR) == "1" diff --git a/src/ucode/smart_routing/session_env.py b/src/ucode/smart_routing/session_env.py index 21887c945..d5f4a467a 100644 --- a/src/ucode/smart_routing/session_env.py +++ b/src/ucode/smart_routing/session_env.py @@ -11,6 +11,7 @@ from ucode.config_io import atomic_write_json from ucode.constants import SMART_ROUTING_ENV_KEYS +from ucode.smart_routing.config import resolve_environment SESSION_ENV_VAR = "UCODE_SESSION_ENV_FILE" SESSION_PYTHON_ENV_VAR = "UCODE_SMART_ROUTER_PYTHON" @@ -53,7 +54,7 @@ def _read(path: Path) -> dict[str, str]: def effective_environment(env: Mapping[str, str] | None = None) -> dict[str, str]: """Overlay the latest session controls on the hook process environment.""" - effective = dict(os.environ if env is None else env) + effective = resolve_environment(env) try: path = session_env_path(effective) except RuntimeError: diff --git a/src/ucode/smart_routing/v2.py b/src/ucode/smart_routing/v2.py index 5e4b1879b..2db12b2fa 100644 --- a/src/ucode/smart_routing/v2.py +++ b/src/ucode/smart_routing/v2.py @@ -31,6 +31,7 @@ ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, LOOPBACK_HOST, + SMART_ROUTER_CONFIG_VERSION_ENV_VAR, SMART_ROUTING_ENV_KEYS, ) from ucode.custom_oauth import custom_oauth_cli_enabled, get_custom_client_token @@ -55,6 +56,7 @@ sync_smart_routing_hooks, ) from ucode.smart_routing.codex_hooks import merge_pre_tool_use_hooks, routing_models +from ucode.smart_routing.config import apply_config, resolve_environment from ucode.smart_routing.session_env import SESSION_ENV_VAR, SESSION_PYTHON_ENV_VAR, start_session from ucode.ui import print_warning @@ -152,8 +154,10 @@ def _model_picker_catalog() -> AnthropicModelCatalog | None: def smart_routing_enabled( env: MutableMapping[str, str] | None = None, *, default: bool = False ) -> bool: - source = os.environ if env is None else env - values = [source.get(var) for var in SMART_ROUTING_ENV_KEYS] + source = resolve_environment(env) + values = [ + source.get(var) for var in (ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR) + ] if "1" in values: return True if "0" in values: @@ -163,7 +167,7 @@ def smart_routing_enabled( def first_prompt_routing_enabled(env: MutableMapping[str, str] | None = None) -> bool: """Whether the first prompt is routed. Subagent-only wins over the full V2 flag.""" - source = os.environ if env is None else env + source = resolve_environment(env) return ( source.get(ENABLE_SMART_ROUTING_ENV_VAR) == "1" and source.get(ENABLE_SUBAGENT_ROUTING_ENV_VAR) != "1" @@ -174,10 +178,7 @@ def enable_smart_routing( env: MutableMapping[str, str] | None = None, ) -> dict[str, str | None]: """Set the full smart-routing env var and return the prior value of every routing var.""" - target = os.environ if env is None else env - previous = {var: target.get(var) for var in SMART_ROUTING_ENV_KEYS} - target[ENABLE_SMART_ROUTING_ENV_VAR] = "1" - return previous + return override_smart_routing(True, env) def override_smart_routing( @@ -187,6 +188,7 @@ def override_smart_routing( """Set an explicit launch-scoped routing choice and return the prior values.""" target = os.environ if env is None else env previous = {var: target.get(var) for var in SMART_ROUTING_ENV_KEYS} + previous.update(apply_config(target)) if enabled: target[ENABLE_SMART_ROUTING_ENV_VAR] = "1" else: @@ -211,7 +213,12 @@ def disable_smart_routing( ) -> dict[str, str | None]: """Temporarily remove the smart-routing env vars and return their prior values.""" target = os.environ if env is None else env - return {var: target.pop(var, None) for var in SMART_ROUTING_ENV_KEYS} + previous = {var: target.pop(var, None) for var in SMART_ROUTING_ENV_KEYS} + if SMART_ROUTER_CONFIG_VERSION_ENV_VAR in target: + previous[SMART_ROUTER_CONFIG_VERSION_ENV_VAR] = target.pop( + SMART_ROUTER_CONFIG_VERSION_ENV_VAR + ) + return previous def _loopback_websocket_url(port: int) -> str: diff --git a/tests/README.md b/tests/README.md index e7cfc9e08..d3b8db57d 100644 --- a/tests/README.md +++ b/tests/README.md @@ -20,11 +20,34 @@ offline tests require GET-only API calls and verify config changes fail without The live fixture compares configuration before and after the journey, even on failure. CUJs never republish configuration or create a remote reservation. CUJ helper tests also verify that unsupported agent names fail rather than defaulting to Codex. +Offline PTY checks verify that the Claude background-task wait observes "No tasks currently +running" or a native task menu containing only completed rows before sending `/exit`, without +stopping tasks or confirming an exit dialog. Running/scheduled sections, incomplete row counts, +and completed-task text outside the native menu cannot satisfy the wait. +The live Claude subagent skill-toggle journey uses this wait after its final calculation. +CUJ4's Claude preset sessions also use it after verifying delegated file tasks: a completed +answer does not prove that a resumed child has stopped running. They cover Claude/Codex helper dispatch and rejection of routing decisions without -the agent-specific prompt-submission evidence. CUJ3 also verifies Claude's native recovery -from the known thinking-display 400: the same payload without display must receive 200. -The smart-routing CUJ runs four fresh sessions: routed and explicit model for both -Claude and Codex. Routing-disabled coverage is deferred until a separately +the agent-specific prompt-submission evidence. +CUJ3 and CUJ4 also verify Claude's native recovery from the known thinking-display 400: +the same payload without display must receive a non-empty 200 for adaptive or enabled thinking. +If the next attempt rejects `safeguards`, only its native removal is accepted before the final 200. +Retries match system context and session metadata so concurrent parent traffic is excluded. +Offline regressions reject missing/failed retries and changes to the model, prompt, budget, or effort. +Delegated tasks wait for the native child's answer, independently of the parent's final turn. +Offline checks require the child's hidden file value; a parent-only answer cannot satisfy the wait. +First-prompt tasks retain completed parent-turn evidence. Offline PTY checks cover billing-notice +input readiness, observed dismissal, and a later notice for child work. +Child HTTP matching accepts Claude's single appended transport newline after the routing +checkpoint; additional text, whitespace, and Codex prompt changes remain rejected. +Offline CUJ regressions cover Codex delegated-turn completion and exact task-request matching; +notifications alone, Claude title requests, and parent continuations do not qualify. +Unrelated and pre-checkpoint request bodies are excluded before JSON decoding. +The original smart-routing CUJ runs four fresh sessions: routed and explicit model for both +Claude and Codex. One additional test has five `SMART_ROUTER_CONFIG_VERSION` cases. +Each case exercises both agents and checks first-prompt routing, orchestrator context in +inference input, and a completed explicitly requested routed subagent. It does not establish +automatic orchestrator delegation. Routing-disabled coverage is deferred until a separately preconfigured workspace is assigned. `test_entry_points.py` also runs both installed console scripts (`ug` and `ucode`) @@ -139,11 +162,31 @@ that Claude settings and Codex's shell policy carry the interpreter and session These are component checks; they do not establish native skill permission matching or PowerShell execution. -The toggle integration journeys run with `ENABLE_SMART_ROUTER_ORCHESTRATOR` unset and with -`ENABLE_SMART_ROUTER_ORCHESTRATOR=1`. They require only `smart-router` by default and both -bundled skills when opted in, verify the saved session controls and native +`test_smart_routing_config.py` includes a 162-case Cartesian component oracle: all three legacy +routing flags take unset, `0`, and `1`, while the selector takes unset or one of the five +supported presets, including the customer first-prompt-and-subagent mode. It uses the named +version constants but independently hardcodes each preset's settings and asserts exact +`resolve_environment` and `apply_config` settings, true-unset omission, unrelated-key and +input preservation, valid-selector consumption, and environment restoration. +Absent/unknown selectors are no-ops; CLI checks cover auth, revert, and launch dispatch. +These component checks do not establish live agent, hook, or gateway behavior. +Three additional component cases compare the legacy flags with `subagent_only_v1`, +`subagent_orch_v1`, and `first_prompt_and_subagent_no_orch_v0`, including first-prompt, +subagent-routing activation, orchestration, and preservation of the chosen router name. + +The toggle integration journeys retain the legacy routing-only and orchestration cases and +add the three subagent-only `SMART_ROUTER_CONFIG_VERSION` presets. They require only +`smart-router` for routing-only cases, and both bundled skills when orchestration is enabled; +they verify the saved +session controls and native tool-result confirmation after each toggle, and explicitly request their children, including while routing is off. +The managed-fixture banner journeys separately cover the selector unset (managed default) and +`first_prompt_and_subagent_no_orch_v0` for real first-prompt tasks. +Preset parameterization augments only routing hooks, skill toggles, and first-prompt routing; +their original legacy-env or managed-default cases remain. Explicit-model selection, command +forwarding, app-server initialization, and catalog fallback retain their original legacy flags +and on/off coverage; preset-specific behavior there is not covered. `test_integration_evidence.py` checks native tool-result extraction for both agents, including collapsed-output records, and excludes user echoes and assistant claims. @@ -195,16 +238,16 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_ug_agents_self_managed_opencode_journey` | Under the injected `managed_workspace_default` config (enables Claude/Codex, not OpenCode): bare configure, `ug agents list`, refused OpenCode launch, `ug agents add opencode`, real headless task, `ug agents remove opencode` (marker `managed_fixture and opencode`, non-blocking CI lane) | OpenCode absent from the list and launch fails with "doesn't enable OpenCode" / `ug agents add opencode`; after add it is listed self-managed and the headless Read task returns the fixture value; after remove it is hidden and refused again | | `test_ug_agents_admin_managed_guardrails` | `ug agents add` / `remove` on an agent the admin config enables | Add is a no-op noting the admin manages it; remove is rejected with nonzero exit | | `test_ug_claude_exports_trace_to_configured_table`, `test_ug_codex_exports_trace_to_configured_table` | Configure tracing, complete a headless task carrying a unique trace marker, then wait for ingestion | The configured trace table contains an agent span with the same trace-safe marker and requested model | -| `test_ug_claude_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` before and after ug's separator, without workspace policy and with routing enabled | Real file task completes; JSON `modelUsage` reports the requested model with output tokens; no routing wrapper | -| `test_ug_codex_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` / `-m VALUE` with routing enabled | Real file task completes; no routing wrapper | +| `test_ug_claude_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` before and after ug's separator with smart routing enabled, without workspace policy | Real file task completes; JSON `modelUsage` reports the requested model with output tokens; no routing wrapper | +| `test_ug_codex_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` / `-m VALUE` with smart routing enabled | Real file task completes; no routing wrapper | | `test_ug_claude_preserves_caller_settings_and_hook` | Pass a settings path containing spaces | Real SessionStart hook executes; caller file unchanged; file task completes | | `test_ug_claude_reports_unsupported_short_model_option` | Pass Claude's unsupported `-m` | Actual agent error and exit status preserved | -| `test_ug_claude_auth_help`, `test_ug_claude_mcp_help` | Request subcommand help, routing off/on | Real agent help; no routing wrapper | -| `test_ug_codex_app_help`, `test_ug_codex_app_server_help`, `test_ug_codex_exec_help`, `test_ug_codex_mcp_help` | Request subcommand help, routing off/on | Real agent help; no routing wrapper | -| `test_ug_codex_app_reports_unknown_argument` | Pass an invalid option directly to `ug codex app`, routing off/on | Real Codex parser error and status preserved | -| `test_ug_codex_app_server_client_initializes` | Connect a stdio client, direct/`--` separator, routing off/on | Actual JSON-RPC initialize response; no non-JSON stdout; no routing | -| `test_smart_routing_claude_route_subagent_hook`, `test_smart_routing_codex_route_subagent_hook` | Pipe a real PreToolUse spawn payload to the installed route-subagent hook with subagent-only routing enabled | Allow decision against the live router; requested model replaced by a routed agent definition (Claude) or bundled catalog slug (Codex) from the offered models; one audited decision matching the session and task | -| `test_smart_router_skill_toggles_claude_subagent_routing`, `test_smart_router_skill_toggles_codex_subagent_routing` | Configure, launch a real subagent-only TUI with orchestration unset or opted in, then spawn tagged children while invoking the installed Smart Router skill to switch routing on -> off -> on in the same session | Only `smart-router` is installed by default; opt-in also installs `smart-router-orchestrator`; all three native children complete; only routing-enabled phases show the subagent banner and produce a live routing decision correlated with the child; no first-prompt routing wrapper; normal exit | +| `test_ug_claude_auth_help`, `test_ug_claude_mcp_help` | Request subcommand help with legacy smart routing off/on | Real agent help; no routing wrapper | +| `test_ug_codex_app_help`, `test_ug_codex_app_server_help`, `test_ug_codex_exec_help`, `test_ug_codex_mcp_help` | Request subcommand help with legacy smart routing off/on | Real agent help; no routing wrapper | +| `test_ug_codex_app_reports_unknown_argument` | Pass an invalid option directly to `ug codex app` with legacy smart routing off/on | Real Codex parser error and status preserved | +| `test_ug_codex_app_server_client_initializes` | Connect a stdio client, direct/`--` separator, with legacy smart routing off/on | Actual JSON-RPC initialize response; no non-JSON stdout; no routing | +| `test_smart_routing_claude_route_subagent_hook`, `test_smart_routing_codex_route_subagent_hook` | Pipe a real PreToolUse spawn payload to the installed route-subagent hook with legacy subagent routing or each supported routing preset | Allow decision against the live router; requested model replaced by a routed agent definition (Claude) or bundled catalog slug (Codex) from the offered models; one audited decision matching the session and task | +| `test_smart_router_skill_toggles_claude_subagent_routing`, `test_smart_router_skill_toggles_codex_subagent_routing` | Configure, launch a real subagent-only TUI with legacy routing-only/orchestration flags or the three subagent-only presets, then spawn tagged children while invoking the installed Smart Router skill to switch routing on -> off -> on in the same session | Routing-only cases install `smart-router`; orchestration cases also install `smart-router-orchestrator`; all three native children complete; only routing-enabled phases show the subagent banner and produce a live routing decision correlated with the child; no first-prompt routing wrapper; the customer full-mode preset is intentionally outside this journey; normal exit | | `test_ug_configure_claude_repeat_and_revert`, `test_ug_configure_codex_repeat_and_revert` | Configure twice over user settings; complete a task; revert twice | Settings preserved; no bearer in ug state; generated config removed; status unconfigured | | `test_ug_configure_claude_cleans_stale_skills_mcp_on_workspace_switch` | Configure the first workspace, register its skills MCP, switch to a second real workspace, and use Claude | Old registration removed from Claude and the new workspace state; old workspace bucket preserved; repeat configure stays clean; real file task completes on the second workspace | | `test_ug_configure_claude_rejects_invalid_credentials`, `test_ug_configure_codex_rejects_invalid_credentials` | Configure with a rejected bearer against the real workspace | Authentication failure; no successful saved setup | @@ -214,7 +257,7 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_case_03_*`, `test_case_05_*` | Pass a provider or model-location override to managed Claude after configure and from fresh state | ug rejects the override before Claude starts and preserves agent-owned state | | `test_case_02_*` | Launch managed Codex after configure and from fresh state | The scoped and stable catalogs, ug-launched app server, and fresh bare app server match the independently fetched admin MPS model IDs. The configured case uses real `ug revert` to remove ug's shared pointer and stable file while preserving a user setting | | `test_case_04_*`, `test_case_06_*` | Pass a provider or model-location override to managed Codex after configure and from fresh state | ug rejects the override before Codex starts and preserves agent-owned state | -| `test_ug_configure_managed_codex_catalog_fallback` | Configure from an injected managed response containing a GPT model absent from Codex's bundled catalog | Actionable metadata warning; conservative catalog entry for the unknown model; real Codex prompt on the valid default model | +| `test_ug_configure_managed_codex_catalog_fallback` | Configure from an injected managed response containing a GPT model absent from Codex's bundled catalog, then inspect the picker with legacy smart routing off/on | Actionable metadata warning; conservative catalog entry for the unknown model; real Codex picker lists the custom catalog model | | `test_managed_fixture_codex_http_headers_in_managed_file` | Interactive PTY configure with injected managed `http_headers` for Codex | The specified header (`x-databricks-workspace`) lands in `model_providers.Databricks.http_headers` in `/etc/codex/managed_config.toml` with the exact admin value | | `test_managed_claude_mps_defaults_accompany_discovery`, `test_managed_claude_parent_schema_defaults_accompany_discovery` | Configure from a stubbed config and launch Claude with MPS discovery (`main.default.ci_e2e_anthropic_mps`) and with `system.ai` Unity Catalog discovery, respectively, both on the managed workspace | Both generated settings files retain every admin-authored default alongside the source header and every independently fetched catalog model with its label; MPS pickers keep family shortcut rows separate from catalog entries; only UC Opus/Sonnet family ids gain `[1m]` | | `test_unmanaged_claude_preserves_preexisting_family_defaults` | Seed Claude's OS-managed family defaults, then configure against one real workspace verified to have no managed config | Every pre-existing Claude family default remains unchanged in the OS-managed settings file | @@ -225,14 +268,14 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_ug_and_ucode_auth_helpers_emit_only_the_supplied_bearer` | Run both auth helper commands with the public bearer override, with and without forced refresh | Exact token-only stdout, no warnings or ANSI escapes; no workspace authentication or saved state | | `test_ug_and_ucode_web_search_helpers_preserve_mcp_stdio` | Initialize and list tools through both web-search helper commands | Exactly the MCP JSON-RPC responses; no text/ANSI contamination; existing server/tool identities preserved; no model request | -With Claude and Codex selected there are **64 live cases** (12 marked TUI cases), +With Claude and Codex selected there are **80 live cases** (14 marked TUI cases), **1 two-workspace case** (marker `workspace_switch`), -**33 managed-fixture cases** (marker `managed_fixture`, with only +**43 managed-fixture cases** (marker `managed_fixture`, with only the CodingAgentConfig input injected from a JSON file in `fixtures/managed_config/`), and **7 installation checks**. The 14 retained numbered scenarios comprise **24 explicit journeys**: 12 managed configured/fresh executions and 12 unmanaged executions. The remaining managed-fixture cases cover focused model, MCP, skills, cache-TTL, and lifecycle shapes, including two Claude defaults cases. Parametrization varies -argument spelling or routing mode, never hides the agent/provider in the test name. Duplicate boot-only cases +argument spelling or external routing selector, never hides the agent/provider in the test name. Duplicate boot-only cases are incorporated into the Databricks configuration TUI journeys. Generated-file cleanup and strict app-server stdout assertions remain enforced. Unmanaged discovery Cases 7–14 configure, list models, or open the picker without @@ -289,7 +332,7 @@ dependency graph to reproduce a user's combination. Every relevant same-reposito PR and push to `main` runs both smoke and the full CUJ suite. Smoke covers the Databricks Hosted configure/TUI, custom OAuth CLI TUI, and headless argument journeys for both agents, in two parallel jobs. After smoke finishes, the full -suite runs all 64 live cases across two parallel agent jobs: one Claude VM and one +suite runs all 80 live cases across two parallel agent jobs: one Claude VM and one Codex VM, each running its configure, headless, and commands/lifecycle cases serially. Each agent is installed once for the full suite, and no two full jobs for the same agent overlap within a run. diff --git a/tests/conftest.py b/tests/conftest.py index c68df98e7..244b5bf2b 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -74,6 +74,8 @@ def reject_privileged_write(path, _desired_text): # header-rendering tests. Clear them so routing stays off unless a test opts in. monkeypatch.delenv("ENABLE_SMART_ROUTING_V2", raising=False) monkeypatch.delenv("ENABLE_SMART_ROUTING_SUBAGENT_ONLY", raising=False) + monkeypatch.delenv("ENABLE_SMART_ROUTER_ORCHESTRATOR", raising=False) + monkeypatch.delenv("SMART_ROUTER_CONFIG_VERSION", raising=False) monkeypatch.delenv("SMART_ROUTER_NAME", raising=False) # On Windows, resolve_command swaps a bare program name for whatever `shutil.which` # finds on the developer's PATH (e.g. a real `codex.CMD`). Rebind only the compatibility diff --git a/tests/e2e_cuj/README.md b/tests/e2e_cuj/README.md index edfb8eafd..3382630a2 100644 --- a/tests/e2e_cuj/README.md +++ b/tests/e2e_cuj/README.md @@ -33,6 +33,11 @@ permission prompts. These checks do not prove the gateway's backing destination. establish coverage. Claude 2.1.290 may receive a 400 rejecting `thinking.display: "updates"`; CUJ3 accepts it only if the next task request removes that field, changes nothing else in the payload, and receives a non-empty HTTP 200. Other failures remain test failures. +The shared recovery check covers both adaptive and enabled thinking, preserving any token budget. +Claude may then receive `safeguards: Extra inputs are not permitted`. Recovery must remove +only that field in the next native attempt and end with a non-empty HTTP 200. The checks +inspect recorded traffic without modifying or replaying requests. +Retries match system context and session metadata to exclude concurrent parent traffic. The test class selects the CUJ3 workspace, `https://dbc-bbdd5508-648e.cloud.databricks.com`. The shared `cuj` fixture supplies its authenticated SDK client and isolated local session; @@ -56,3 +61,59 @@ Collection only (no authentication or inference): uv run pytest -c tests/e2e_cuj/pytest.ini --confcutdir=tests/e2e_cuj \ --collect-only tests/e2e_cuj ``` + +## Smart-routing CUJ + +`test_cuj4_smart_routing.py` uses its own read-only workspace with managed smart routing enabled. +It runs the original routed/explicit-model journeys and five +`SMART_ROUTER_CONFIG_VERSION` cases using the shared version constants; each launches Claude +and Codex once. +The independent expectation table is: + +| Selector | First prompt routed | Orchestrator context | Child on first prompt | Explicitly requested child | +| --- | --- | --- | --- | --- | +| `first_prompt_and_subagent_no_orch_v0` | Yes | No | No | Yes, routed | +| `subagent_only_v0` | No | No | No | Yes, routed | +| `subagent_only_v1` | No | No | No | Yes, routed | +| `subagent_orch_v0` | No | Yes | No | Yes, routed | +| `subagent_orch_v1` | No | Yes | No | Yes, routed | + +First prompts explicitly forbid delegation to isolate first-prompt routing. Assertions require +the expected presence/absence of a prompt-correlated router request, successful inference on +the routed or configured default model, completed native file-task evidence, and no child session. +Orchestrator presence means its activation context reached the real gateway inference input, +not that an assistant echoed it or a skill merely existed on disk. +Task inference must contain the exact task/routed prompt and tools, excluding Claude title +requests and parent continuations from child-inference checks. Only Claude child requests +after the routing checkpoint may append one transport newline to that exact prompt. +Parent and child requests use the same verified thinking-display recovery as CUJ3: +only known display/safeguards rejections followed by native removal of the rejected field +and a final non-empty 200 are accepted. Model, prompt, budget, and effort must remain unchanged. +Each preset session then explicitly requests one child for a separate hidden-value +file task. Assertions require a new native child transcript containing the value, +a correlated spawn-routing decision, and successful child inference on the +router's selected model. This tests requested delegation, not automatic orchestrator delegation. +The wait uses the child's answer directly; parent-answer reconstruction is reserved for first-prompt tasks. +Offline checks reject parent-only answers and accept child answers before the parent replies. +Offline evidence regressions do not establish a live CUJ pass. +Selector cases have separate TUI artifact names, and the session environment is restored afterward. +Claude sessions start in auto mode; unless a routed first prompt switches the session to Haiku +(which leaves it in manual mode), its classifier requests through the recording proxy trigger +Claude's informational auto-mode classifier billing notice over the transcript. The CUJ terminal +waits for that exact notice to render stably, presses Enter (continue), and observes dismissal +before continuing. Later occurrences are handled the same way; any other dialog still fails. +After verifying the delegated task, Claude sessions also wait in the native `/tasks` view +until no background work remains, including a child resumed after its initial answer. +Only then does the test close the task view and submit `/exit`. +Each preset also runs public `ug revert` in cleanup, including after a failed assertion, so +interactive launches' OS-managed settings cannot contaminate the next preset's configuration. +The existing Claude explicit-model precedence case remains skipped; the routing-disabled case +still requires a separately preconfigured workspace. + +With the same live prerequisites: + +```bash +uv run --with pexpect==4.9.0 --with pyte==0.8.2 pytest \ + --confcutdir=tests/e2e_cuj tests/e2e_cuj/test_cuj4_smart_routing.py \ + -k test_smart_router_config_version -v +``` diff --git a/tests/e2e_cuj/helpers/evidence.py b/tests/e2e_cuj/helpers/evidence.py index 9f580d567..73a401474 100644 --- a/tests/e2e_cuj/helpers/evidence.py +++ b/tests/e2e_cuj/helpers/evidence.py @@ -2,6 +2,7 @@ from __future__ import annotations +import json import re from abc import ABC, abstractmethod from dataclasses import asdict, dataclass @@ -48,6 +49,71 @@ def assert_served(recorder, request, model): assert response.body, "Inference response was empty" +def served_inference_request(recorder, requests, request, agent): + """Require HTTP 200 or verified native Claude compatibility retries.""" + model = request.payload["model"] + # Claude 2.1.290 may remove display, then safeguards after separate Bedrock + # validation errors. Inspect recorded traffic only; never replay or edit it. + # Each field may disappear once, and the full task/model/budget/effort must survive. + for _ in range(3): + payload = request.payload + thinking = payload.get("thinking", {}) + thinking_type = thinking.get("type") + response = recorder.response_for(request, timeout=240) + error_message = None + if agent == CLAUDE and response.status_code == 400: + try: + error = httpx.Response( + response.status_code, headers=response.headers, content=response.body + ).json() + if isinstance(error, dict) and error.get("error_code") == "BAD_REQUEST": + detail = json.loads(error.get("message", "")) + if isinstance(detail, dict) and set(detail) == {"message"}: + error_message = detail["message"] + except (ValueError, TypeError): + pass + expected_retry = None + if ( + thinking_type in {"adaptive", "enabled"} + and thinking.get("display") == "updates" + and error_message + == f"thinking.{thinking_type}.display: Input should be 'summarized', 'omitted'" + ): + expected_retry = { + **payload, + "thinking": {key: value for key, value in thinking.items() if key != "display"}, + } + elif ( + "safeguards" in payload + and error_message == "safeguards: Extra inputs are not permitted" + ): + expected_retry = {key: value for key, value in payload.items() if key != "safeguards"} + if expected_retry is None: + assert_served(recorder, request, model) + return request + following = requests[requests.index(request) + 1 :] + # Parent and child traffic can interleave on the same endpoint. + retry = next( + ( + candidate + for candidate in following + if candidate.method == request.method + and candidate.path == request.path + and candidate.payload.get("system") == payload.get("system") + and candidate.payload.get("metadata") == payload.get("metadata") + ), + None, + ) + assert retry is not None, "Claude compatibility rejection had no retry" + assert retry.payload == expected_retry, ( + "Claude compatibility retry changed more than the rejected field", + error_message, + retry.payload, + ) + request = retry + raise AssertionError("Claude compatibility retries did not reach a successful response") + + def claude_file_task(session): """A FileTask naming its absolute path, so Claude reads it without a `find` permission prompt.""" task = FileTask(session) @@ -156,48 +222,77 @@ def _completed_turn(records, task): meta = [row["payload"] for row in records if row.get("type") == "session_meta"] if len(meta) != 1 or isinstance(meta[0].get("source"), dict): return None - contexts, prompt_count, seen_prompt, active_turn = {}, 0, False, None + contexts, active_turn, target_turn = {}, None, None + started_turn, aborted_turn, answer, prompt_sources = None, None, None, set() for row in records: payload = row.get("payload", {}) + metadata = row.get("internal_chat_message_metadata_passthrough") or {} + if "multi_agent.subagent_notification" in metadata.get( + "content_item_kinds", () + ) or "" in str( + payload.get("content") or payload.get("message") + ).replace("-", "_"): + continue + row_turn = payload.get("turn_id") if row.get("type") == "turn_context": - contexts.setdefault(payload.get("turn_id"), []).append(payload.get("model")) - if ( + contexts.setdefault(row_turn, []).append(payload.get("model")) + if target_turn and row_turn != target_turn: + return None + active_turn = row_turn + continue + source = None + if row.get("type") == "event_msg" and payload.get("type") == "user_message": + source, prompt = "event", payload.get("message") + elif ( row.get("type") == "response_item" and payload.get("type") == "message" and payload.get("role") == "user" ): - prompt = message_text(payload.get("content")) + source, prompt = "response", message_text(payload.get("content")) + if source: if prompt == task.prompt: - prompt_count += 1 - assert prompt_count == 1, "Prompt was submitted more than once" - seen_prompt = True - elif seen_prompt: + assert source not in prompt_sources, "Prompt was submitted more than once" + prompt_sources.add(source) + if target_turn and row_turn and row_turn != target_turn: + return None + target_turn = target_turn or row_turn or active_turn + elif prompt_sources: return None + continue if row.get("type") != "event_msg": continue - if payload.get("type") == "task_started": - if seen_prompt: + event_type = payload.get("type") + if event_type == "task_started": + turn_id = payload.get("turn_id") + if not turn_id or turn_id == aborted_turn: return None - active_turn = payload.get("turn_id") - if payload.get("type") == "user_message": - if payload.get("message") == task.prompt: - prompt_count += 1 - assert prompt_count == 1, "Prompt was submitted more than once" - seen_prompt = True - elif seen_prompt: + if target_turn and turn_id != target_turn: return None - if seen_prompt and payload.get("type") == "task_complete": + active_turn = started_turn = turn_id + if prompt_sources and target_turn is None: + target_turn = turn_id + elif event_type == "task_complete": turn_id = payload.get("turn_id") - answer = payload.get("last_agent_message") or "" - if ( - task.value in answer - and turn_id - and turn_id == active_turn - and contexts.get(turn_id) - ): - assert all(contexts[turn_id]), "Missing native turn model metadata" - return CompletedTurn(meta[0]["id"], turn_id, contexts[turn_id], answer) - return None + if target_turn: + if turn_id != target_turn: + return None + answer = payload.get("last_agent_message") or "" + break + if turn_id == active_turn: + active_turn = started_turn = None + elif event_type == "turn_aborted": + aborted_turn = payload.get("turn_id") + if aborted_turn == target_turn: + return None + if aborted_turn == active_turn: + active_turn = started_turn = None + if not prompt_sources or not target_turn or started_turn != target_turn or answer is None: + return None + models = contexts.get(target_turn) + if task.value not in answer or not models: + return None + assert all(models), "Missing native turn model metadata" + return CompletedTurn(meta[0]["id"], target_turn, models, answer) def get_cuj_helper(agent): diff --git a/tests/e2e_cuj/helpers/terminal.py b/tests/e2e_cuj/helpers/terminal.py index 3ef58715a..044b8c8ae 100644 --- a/tests/e2e_cuj/helpers/terminal.py +++ b/tests/e2e_cuj/helpers/terminal.py @@ -6,6 +6,14 @@ from .constants import CLAUDE, CODEX +def _shows_auto_mode_billing_notice(screen): + return ( + "We're changing auto mode to no longer charge for classifier requests" in screen + and "Nothing breaks: auto mode keeps working" in screen + and "Enter to continue · Esc to cancel" in screen + ) + + class Terminal(AgentTerminal): def __init__(self, session, name, args, *, agent=None): """Run `ug `; bare `ug` (empty args) names the agent it is expected to launch.""" @@ -26,6 +34,7 @@ def wait_until( """Wait for `done()`, failing on API errors or any `rejected` permission prompt. `on_screen(screen)` may answer an expected dialog; it returns True when it sent keys. + Claude's auto-mode classifier billing notice is acknowledged with Enter. """ def completed(screen): @@ -34,6 +43,21 @@ def completed(screen): assert not any(prompt in screen for prompt in rejected), ( "Unexpected permission request; inspect the actual command:\n" + screen ) + if _shows_auto_mode_billing_notice(screen): + # Auto-mode classifier requests through the recording proxy raise this + # informational modal over the transcript. Wait for its input handler + # before pressing Enter, then observe dismissal before continuing. + self.wait_for( + _shows_auto_mode_billing_notice, + "the auto-mode classifier billing notice", + stable_for=0.5, + ) + self.send("\r", "acknowledge auto-mode classifier billing notice") + self.wait_for( + lambda text: not _shows_auto_mode_billing_notice(text), + "dismissal of the auto-mode classifier billing notice", + ) + return False if on_screen is not None and on_screen(screen): return False return done() diff --git a/tests/e2e_cuj/test_cuj3_models.py b/tests/e2e_cuj/test_cuj3_models.py index 7eb807210..d427cbf5f 100644 --- a/tests/e2e_cuj/test_cuj3_models.py +++ b/tests/e2e_cuj/test_cuj3_models.py @@ -2,7 +2,6 @@ import json -import httpx import pytest from tests.integration.utils.agents import claude, codex @@ -27,7 +26,13 @@ OTHER_MODEL_SCHEMA, ) from .helpers.constants import CLAUDE, CODEX, INFERENCE_PATHS -from .helpers.evidence import SessionEvidence, assert_served, claude_file_task, message_text +from .helpers.evidence import ( + SessionEvidence, + assert_served, + claude_file_task, + message_text, + served_inference_request, +) from .helpers.terminal import Terminal CUJ_NAME = "CUJ 3 · UC model discovery" @@ -105,23 +110,6 @@ def _request_contains_task(request, agent, task): return False -def _is_thinking_display_rejection(response): - """Recognize the observed gateway rejection, including compressed error bodies.""" - if response.status_code != 400: - return False - try: - error = httpx.Response( - response.status_code, headers=response.headers, content=response.body - ).json() - if not isinstance(error, dict) or error.get("error_code") != "BAD_REQUEST": - return False - return json.loads(error.get("message", "")) == { - "message": "thinking.adaptive.display: Input should be 'summarized', 'omitted'" - } - except (ValueError, TypeError): - return False - - def _assert_inference_evidence(recorder, checkpoint, agent, task, expected): expected_wire_model = claude.discovery_model_id(expected) if agent == CLAUDE else expected requests = recorder.requests_after(checkpoint) @@ -139,29 +127,9 @@ def _assert_inference_evidence(recorder, checkpoint, agent, task, expected): request for request in inference_requests if _request_contains_task(request, agent, task) ] assert task_requests, "No inference request contained the submitted task prompt" - # Claude 2.1.290 sends thinking.display="updates" to custom endpoints; - # 2.1.280 restricted it to Anthropic's first-party base URL. This workspace - # rejects "updates" with 400. Claude retries without it and completes with - # effort="high" unchanged. Accept only this rejection paired with a successful - # retry of the same payload with display removed; all other requests must return 200. - # Live A/B: https://github.com/databricks/unity-gateway/actions/runs/37865757991 - # TODO: Remove this workaround once we add thinking-display-updates-2026-08-18 - # to accepted betas on Bedrock passthrough. - for index, request in enumerate(task_requests): - if ( - agent == CLAUDE - and request.payload.get("thinking") == {"type": "adaptive", "display": "updates"} - and _is_thinking_display_rejection(recorder.response_for(request, timeout=240)) - ): - assert index + 1 < len(task_requests), "Thinking display rejection had no retry" - retry = task_requests[index + 1] - assert retry.payload == {**request.payload, "thinking": {"type": "adaptive"}}, ( - "Thinking display retry changed more than display", - retry.payload, - ) - assert_served(recorder, retry, expected_wire_model) - continue - assert_served(recorder, request, expected_wire_model) + for request in task_requests: + served = served_inference_request(recorder, task_requests, request, agent) + assert_served(recorder, served, expected_wire_model) @pytest.fixture(autouse=True) diff --git a/tests/e2e_cuj/test_cuj4_smart_routing.py b/tests/e2e_cuj/test_cuj4_smart_routing.py index 2c6a56423..c19ab368e 100644 --- a/tests/e2e_cuj/test_cuj4_smart_routing.py +++ b/tests/e2e_cuj/test_cuj4_smart_routing.py @@ -4,7 +4,21 @@ import pytest -from tests.integration.utils.evidence import FileTask +from tests.integration.utils.evidence import ( + FileTask, + agent_sessions, + assert_subagent_routed, + assistant_answers, + is_child_session, + read_jsonl, +) +from ucode.smart_routing.config import ( + FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0, + SUBAGENT_ONLY_V0, + SUBAGENT_ONLY_V1, + SUBAGENT_ORCH_V0, + SUBAGENT_ORCH_V1, +) from .base import BaseCujTest from .helpers.constants import CLAUDE, CODEX, INFERENCE_PATHS, CodingAgent @@ -13,6 +27,7 @@ SessionObservation, canonical_model, claude_file_task, + served_inference_request, ) from .helpers.terminal import Terminal from .helpers.tui_request_recorder import RecordedRequest, RecordedResponse @@ -22,6 +37,7 @@ ROUTING_PATH = "/ai-gateway/routing/v1/routes:select" AGENTS = (CLAUDE, CODEX) +ORCHESTRATOR_CONTEXT = "Smart Router Orchestrator is on for this session." @dataclass(frozen=True) @@ -71,6 +87,47 @@ def _run_session(session, recorder, agent, task, launch_args): return evidence.observe(task), recorder.requests_after(checkpoint) +def _task_inference_request(requests, agent, prompt, *, after=0): + """Match the task payload, allowing Claude's single child-prompt newline.""" + prompts = {prompt} + if agent == CLAUDE and after: + prompts.add(prompt + "\n") + for request in requests: + if ( + request.sequence <= after + or request.method != "POST" + or request.path != INFERENCE_PATHS[agent] + ): + continue + payload = request.payload + tools = payload.get("tools") + if not isinstance(tools, list) or not tools: + continue + entries = payload.get("messages" if agent == CLAUDE else "input", []) + if isinstance(entries, str): + if entries in prompts: + return request + continue + if not isinstance(entries, list): + continue + for entry in entries: + if not isinstance(entry, dict) or entry.get("role") != "user": + continue + content = entry.get("content", []) + if isinstance(content, str): + if content in prompts: + return request + elif isinstance(content, list) and any( + isinstance(part, dict) + and part.get("type") in {"text", "input_text"} + and isinstance(part.get("text"), str) + and part.get("text") in prompts + for part in content + ): + return request + raise AssertionError("No tool-capable inference request contained the exact task prompt") + + def _assert_published_config_matches_expectations(published): assert published["spec_version"] == 1 assert published["default_agent"] == CodingAgent.CLAUDE_CODE @@ -125,11 +182,8 @@ def run_smart_routing_journeys(cuj) -> SmartRoutingSessionResults: if request.method == "POST" and request.path == ROUTING_PATH ) route_response = recorder.response_for(route_request) - inference_request = next( - request - for request in requests - if request.method == "POST" and request.path == INFERENCE_PATHS[agent] - ) + inference_request = _task_inference_request(requests, agent, task.prompt) + inference_request = served_inference_request(recorder, requests, inference_request, agent) no_model_override[agent] = SessionCase( agent=agent, launch_args=(agent,), @@ -145,11 +199,8 @@ def run_smart_routing_journeys(cuj) -> SmartRoutingSessionResults: task = _file_task(session, agent) launch_args = (agent, "--model", overrides[agent]) observation, requests = _run_session(session, recorder, agent, task, launch_args) - inference_request = next( - request - for request in requests - if request.method == "POST" and request.path == INFERENCE_PATHS[agent] - ) + inference_request = _task_inference_request(requests, agent, task.prompt) + inference_request = served_inference_request(recorder, requests, inference_request, agent) with_model_override[agent] = SessionCase( agent=agent, launch_args=launch_args, @@ -178,6 +229,133 @@ def completed_smart_routing_runs(cuj): class TestCujSmartRouting(BaseCujTest): WORKSPACE_URL = "https://dbc-1a9622fc-2e91.cloud.databricks.com/" + @pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION, first_prompt_routed, orchestrator_enabled", + [ + (FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0, True, False), + (SUBAGENT_ONLY_V0, False, False), + (SUBAGENT_ONLY_V1, False, False), + (SUBAGENT_ORCH_V0, False, True), + (SUBAGENT_ORCH_V1, False, True), + ], + ) + def test_smart_router_config_version( + self, cuj, SMART_ROUTER_CONFIG_VERSION, first_prompt_routed, orchestrator_enabled + ): + """Scenario: launch both agents with a preset, then explicitly request a subagent. + + Expected: first-prompt routing and orchestrator context match the preset; + one routed native child completes the delegated task. + """ + session, workspace, recorder = cuj + previous = session.env.get("SMART_ROUTER_CONFIG_VERSION") + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION + try: + configs = _assert_published_config_matches_expectations(workspace.config()) + recorder.configure_session(session, ["configure", "--disable-databricks-ai-tools"]) + for agent in AGENTS: + supported = workspace.model_ids(agent) + evidence = SessionEvidence(session.home, agent) + existing_sessions = set(agent_sessions(session, agent)) + task = FileTask(session) + task.prompt += " Do not delegate." + checkpoint = recorder.checkpoint() + recorder.prepare_launch() + with Terminal(session, f"{SMART_ROUTER_CONFIG_VERSION}-{agent}", [agent]) as tui: + tui.boot(timeout=150) + tui.submit(task.prompt) + tui.task(evidence, task) + requests = recorder.requests_after(checkpoint) + routes = [request for request in requests if request.path == ROUTING_PATH] + assert len(routes) == int(first_prompt_routed), (agent, routes) + expected_model = configs[agent]["default_models"]["default_model"] + if first_prompt_routed: + assert routes[0].payload["task"]["prompt"] == task.prompt + response = recorder.response_for(routes[0]) + assert response.status_code == 200 + selections = response.payload["route_selection"] + assert len(selections) == 1 + expected_model = selections[0]["route_option"]["model"] + inference = _task_inference_request(requests, agent, task.prompt) + inference = served_inference_request(recorder, requests, inference, agent) + assert canonical_model(inference.payload["model"]) == canonical_model( + expected_model + ) + evidence.assert_applied(task, supported, expected=expected_model) + assert ( + ORCHESTRATOR_CONTEXT.encode() in inference.body + ) == orchestrator_enabled, agent + assert not any( + is_child_session(agent, path, records) + for path, records in agent_sessions(session, agent).items() + if path not in existing_sessions + ), agent + + child_task = FileTask(session) + child_task.prompt = child_task.delegate_prompt + decisions_path = ( + session.home / ".ucode" / f"{agent}-smart-routing-decisions.jsonl" + ) + decision_count = len(read_jsonl(decisions_path)) + checkpoint = recorder.checkpoint() + existing_sessions = set(agent_sessions(session, agent)) + tui.submit(child_task.prompt) + tui.wait_until( + lambda task=child_task, agent=agent: task.completed( + session, agent, child=True + ), + "completed native subagent file task", + ) + children = { + path: records + for path, records in agent_sessions(session, agent).items() + if path not in existing_sessions and is_child_session(agent, path, records) + } + assert len(children) == 1, (agent, children.keys()) + assert any( + child_task.value in answer + for records in children.values() + for answer in assistant_answers(agent, records) + ), agent + decisions = read_jsonl(decisions_path)[decision_count:] + assert_subagent_routed( + session, + agent, + child_task, + decision_ids={decision["decision_id"] for decision in decisions}, + ) + requests = recorder.requests_after(checkpoint) + routes = [request for request in requests if request.path == ROUTING_PATH] + assert len(routes) == 1, (agent, routes) + route_prompt = routes[0].payload["task"]["prompt"] + assert child_task.filename in route_prompt + response = recorder.response_for(routes[0]) + assert response.status_code == 200 + selections = response.payload["route_selection"] + assert len(selections) == 1 + inference = _task_inference_request( + requests, + agent, + route_prompt, + after=routes[0].sequence, + ) + inference = served_inference_request(recorder, requests, inference, agent) + assert canonical_model(inference.payload["model"]) == canonical_model( + selections[0]["route_option"]["model"] + ) + if agent == CLAUDE: + tui.wait_for_background_tasks() + tui.exit_normally() + finally: + if previous is None: + session.env.pop("SMART_ROUTER_CONFIG_VERSION", None) + else: + session.env["SMART_ROUTER_CONFIG_VERSION"] = previous + session.revert_machine_wide( + f"{SMART_ROUTER_CONFIG_VERSION}-revert", + "Smart-router preset cleanup left machine-wide agent settings", + ) + @pytest.mark.parametrize("agent", AGENTS) def test_agent_completes_real_first_prompt_file_task_without_model_override( self, completed_smart_routing_runs, agent diff --git a/tests/integration/README.md b/tests/integration/README.md index 055b00853..8f9171276 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -10,6 +10,28 @@ CUJ2 adds three separately collected cases for exact published MPS/MCP configura inference, and Claude inference. The config equality also accounts for the workspace's fixture skill name as data; skill download and invocation are not covered. +The dedicated smart-routing CUJ in `../e2e_cuj/test_cuj4_smart_routing.py` leaves the original +managed-default and explicit-model cases unchanged. One additional test runs the five supported +`SMART_ROUTER_CONFIG_VERSION` values, checking both agents' first-prompt routing, orchestrator +context in inference input, and completed explicitly requested routed subagents. +This is not automatic orchestrator-delegation coverage. No workspace configuration is modified. +CUJ3 and CUJ4 require a non-empty HTTP 200 for inference. Claude 2.1.290's known +thinking-display 400 is accepted only when the next inference request with the same system +context and session metadata removes `display`, +preserves every other payload field, and succeeds; adaptive and enabled thinking are covered. +The native retry may also encounter `safeguards: Extra inputs are not permitted`; that +requires the next recorded attempt to remove only `safeguards` and reach a non-empty 200. +Offline regressions check both CUJs and reject changes to the model, prompt, budget, or effort. +Each preset cleans up interactive OS-managed settings with public `ug revert`, even on failure. +Delegated CUJ tasks wait for the native child's hidden file value, independently of the parent's +final turn. Offline checks reject parent-only answers. First-prompt tasks retain parent-turn evidence. +The CUJ terminal waits for stable billing-notice rendering before Enter and observes dismissal; +an offline PTY regression also exercises a later notice during child work. +Claude child HTTP prompts may carry one appended newline after the route-selection checkpoint; +offline cases reject any other prompt changes and retain exact Codex matching. +Offline CUJ cases also cover Codex delegated-turn completion and task-request matching; +see `../e2e_cuj/README.md` for the evidence requirements. + The [catalog discovery journey](../e2e_cuj/README.md) uses the CUJ3 workspace to check agent-compatible pickers, schema exclusions, configured defaults, and real inference. CI collects it through the shared `dedicated-cuj` job. Its request checks accept the known @@ -23,6 +45,17 @@ or construct ug state files. The normal test suite checks these boundaries. The existing unit tests keep their fixtures. Integration has an independent pytest configuration and uses `--confcutdir` so those fixtures cannot leak in. It is not collected by the default `uv run pytest` command. +The Claude subagent skill-toggle journey opens `/tasks` after its final calculation +and waits up to 180 seconds for "No tasks currently running" or a native menu containing only +completed task rows before submitting `/exit`. Claude 2.1.290 retains finished children in a +`Completed (N)` list, so the empty-state message is not required when the list has exactly N +checkmarked `done` rows and no running or scheduled section. The helper observes menu dismissal +after Escape before typing `/exit`. Offline PTY checks reject active/scheduled tasks, partial +lists, and completed-task text outside the native menu. +It neither stops tasks nor confirms an exit dialog. Process exit retains its 30-second +timeout. Offline PTY checks cover this ordering. +The shared wait also runs before CUJ4's Claude preset sessions exit, after delegated-task +evidence is checked, because Claude can resume a child after its first completed answer. `TestChildStdoutLaunch` in `../test_cli.py` covers clean Claude print-mode and Codex exec/app-server stdout, early launch errors, and forwarding through ug's `--`. @@ -291,10 +324,11 @@ PATH conflicts for the Smart Router skill have subprocess/component coverage in `ug` first in PATH. The live journeys above do not inject a second installation or establish PowerShell command execution. -The toggle journeys run with `ENABLE_SMART_ROUTER_ORCHESTRATOR` unset and with -`ENABLE_SMART_ROUTER_ORCHESTRATOR=1`. They require only `smart-router` by default and both -`smart-router-orchestrator` and `smart-router` when opted in. They verify the saved session -controls, a new CLI confirmation in the native tool-result records, and a new +The toggle journeys retain the legacy routing-only and orchestration cases and add each of +the three subagent-only `SMART_ROUTER_CONFIG_VERSION` presets. They require only `smart-router` +for routing-only cases, and both `smart-router-orchestrator` and `smart-router` when orchestration +is enabled. They verify the saved session controls, a new CLI confirmation in the native +tool-result records, and a new assistant answer after each skill invocation. Collapsed terminal output is allowed; the answer need not repeat the CLI's exact wording. Each following child still verifies whether a routing decision occurred. @@ -305,6 +339,23 @@ role-contract preservation, and isolation from legacy preference files lack dedicated regression coverage. Codex's native hook merging, project trust, and execution of pre-existing hooks are not exercised by this integration suite. +Preset parameterization augments only routing hooks, skill toggles, and first-prompt routing; +their original legacy-env or managed-default cases remain. Explicit-model selection, command +forwarding, app-server initialization, and catalog fallback retain their original legacy flags +and on/off coverage; preset-specific behavior there is not covered. + +The unit/component `../test_smart_routing_config.py` includes a 162-case Cartesian oracle over all +three legacy routing flags (`None`, `0`, `1`) and six selector forms (`None` plus the five +supported presets, including the customer first-prompt-and-subagent mode). It independently +hardcodes preset values and asserts exact +`resolve_environment` and `apply_config` settings, true-unset omission, unrelated-key and +input preservation, valid-selector consumption, and environment restoration. +Absent/unknown selectors are no-ops; CLI checks cover auth, revert, and launch dispatch. +These component checks do not establish live agent, hook, or gateway behavior. +Three component cases additionally compare the equivalent legacy flags and presets for +`subagent_only_v1`, `subagent_orch_v1`, and `first_prompt_and_subagent_no_orch_v0`; +they verify routing activation, first-prompt behavior, orchestration, and router-name preservation. + The portable `../test_claude_windows_smart_routing.py` checks the Windows subagent-only fallback without Unix imports. Native Windows TUI and hook execution remain outside this integration suite. @@ -386,10 +437,10 @@ startup banners and footer text cannot satisfy discovery assertions. Cases 7–1 they only configure, list models, and open/close the picker. Other live CUJs perform real model tasks. -There are **64 live cases** (including 12 marked TUI journeys) and **7 installation +There are **80 live cases** (including 14 marked TUI journeys) and **7 installation checks** with Claude and Codex; selecting OpenCode adds one live headless case. One **`workspace_switch` case** uses two real workspaces and checks skills MCP cleanup and a completed -Claude task. A further **33 `managed_fixture` cases** (two of them also `live`) run on the +Claude task. A further **43 `managed_fixture` cases** (ten of them also `live`) run on the managed workspace with a checked-in JSON CodingAgentConfig from `tests/fixtures/managed_config/` injected through `UCODE_MANAGED_CONFIG_STUB`; there are no cases that read a published config. Twelve explicit configured/fresh Claude and Codex discovery and source-override journeys use the @@ -406,7 +457,7 @@ catalog discovery with overall defaults, family defaults, or both, along with ex selection and preservation of static model lists. Both cases run on the managed workspace's own bearer; no second workspace or extra secret is involved. The 14 retained numbered scenarios comprise 24 explicit journeys: 12 managed and 12 unmanaged -executions; the complete integration suite collects 103 executions. See the named coverage and gaps matrix in +executions; the complete integration suite collects 121 executions. See the named coverage and gaps matrix in [../README.md](../README.md). ```bash @@ -549,13 +600,13 @@ each test; only explicit-model scenarios choose and record a discovered Every same-repository PR and push to `main` runs **Smoke journeys**, followed by **Full journeys** even if smoke fails. Smoke runs the Hosted configure/TUI, headless argument, and custom OAuth CLI TUI journeys for each agent (six cases, -two agent jobs). Full runs all 64 live cases, including those smoke cases, in two +two agent jobs). Full runs all 80 live cases, including those smoke cases, in two disjoint agent lanes: | Agent lane | Marker | Cases | | --- | --- | --- | -| Claude | `live and claude` | 30 | -| Codex | `live and codex` | 34 | +| Claude | `live and claude` | 38 | +| Codex | `live and codex` | 42 | A non-blocking **OpenCode** job (`live and opencode`, one case) runs alongside them with `continue-on-error` and is not part of the required `cujs` gate until it is stable. @@ -633,15 +684,15 @@ Codex state comparisons exclude `.codex/tmp/arg0`, the disposable executable lin recreated by version checks, while continuing to compare persistent agent files. In addition, `test_ug_configure_managed_codex_catalog_fallback` injects the intentionally nonexistent `system.ai.gpt-99`, keeping it out of the real workspace while launching Codex through that -workspace on the valid default model `system.ai.gpt-5-6-sol`. With smart routing enabled, it opens -the real Codex `/models` picker and requires that injected custom-catalog model to be listed. The -same picker assertion also runs with smart routing disabled to cover both launch paths. +workspace on the valid default model `system.ai.gpt-5-6-sol`. With legacy smart routing off and +on, it opens the real Codex `/models` picker and requires that injected custom-catalog model +to be listed. The fixture itself has no managed smart-routing setting. The smart-routing banner journeys inject static Claude and Codex model lists with `smart_routing` enabled in the agent config, run `ug configure`, then launch the real TUI and -submit one small file task. Each asserts the "Using Unity Gateway Smart Router." banner naming -the selected model appears in the TUI, the routed answer completes the file task, and the -session exits normally. +submit one small file task. Each runs once with the selector unset (managed default) and once +with `first_prompt_and_subagent_no_orch_v0`, asserting the "Using Unity Gateway Smart Router." +banner naming the selected model, a routed answer completing the file task, and normal exit. That workspace authenticates as a service principal, so CI mints a short-lived token per run from these same-repository secrets rather than storing a long-lived bearer: @@ -864,7 +915,7 @@ uv run --no-project --python 3.12 python scripts/run_integration.py \ unset DATABRICKS_BEARER ``` -This runs all 64 live cases. For the seven installation checks, run the same +This runs all 80 live cases. For the seven installation checks, run the same runner/version/index arguments with `--installation-only` and omit `-- -m live`; no bearer or workspace is needed. Results remain under `.integration-runs/`. Each invocation needs a new output directory; an existing one is rejected. diff --git a/tests/integration/test_ug_configure_managed_models.py b/tests/integration/test_ug_configure_managed_models.py index bf2b03410..c8ba5b9ab 100644 --- a/tests/integration/test_ug_configure_managed_models.py +++ b/tests/integration/test_ug_configure_managed_models.py @@ -221,22 +221,35 @@ def test_ug_configure_managed_codex_catalog_fallback(live_session, workspace, ro @pytest.mark.managed_fixture @pytest.mark.claude -def test_managed_fixture_claude_smart_routing_banner(live_session, workspace): - """Scenario: an admin config lists Claude models and enables smart routing for Claude. - - Expected: `ug configure` applies the config without the personal agent selector, and - `ug claude` routes the first real prompt: the TUI shows the Unity Gateway Smart Router - banner naming the selected model, the routed answer completes the file task, and the - session exits normally. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [None, "first_prompt_and_subagent_no_orch_v0"], + ids=["managed-default", "first_prompt_and_subagent_no_orch_v0"], +) +def test_managed_fixture_claude_smart_routing_banner( + live_session, workspace, SMART_ROUTER_CONFIG_VERSION +): + """Scenario: an admin config lists Claude models and launch uses its default or the + customer first-prompt-and-subagent selector. + + Expected: `ug configure` applies the config without the personal agent selector, and both + launch modes route the first real prompt: the TUI shows the Unity Gateway Smart Router + banner naming the selected model, the routed answer completes the file task, and the session + exits normally. """ session = live_session task = FileTask(session) use_managed_config_fixture(session, "claude_smart_routing") result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) assert "Select coding agents to configure:" not in result.stdout, result.stdout + if SMART_ROUTER_CONFIG_VERSION is not None: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION with AgentTerminal( - session, "claude", [str(session.binary), "claude"], "managed-smart-routing" + session, + "claude", + [str(session.binary), "claude"], + f"managed-smart-routing-{SMART_ROUTER_CONFIG_VERSION or 'managed-default'}", ) as tui: tui.boot() tui.submit(task.prompt) @@ -253,22 +266,35 @@ def test_managed_fixture_claude_smart_routing_banner(live_session, workspace): @pytest.mark.managed_fixture @pytest.mark.codex -def test_managed_fixture_codex_smart_routing_banner(live_session, workspace): - """Scenario: an admin config lists Codex models and enables smart routing for Codex. - - Expected: `ug configure` applies the config without the personal agent selector, and - `ug codex` routes the first real prompt: the TUI shows the Unity Gateway Smart Router - banner naming the selected model, the routed answer completes the file task, and the - session exits normally. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [None, "first_prompt_and_subagent_no_orch_v0"], + ids=["managed-default", "first_prompt_and_subagent_no_orch_v0"], +) +def test_managed_fixture_codex_smart_routing_banner( + live_session, workspace, SMART_ROUTER_CONFIG_VERSION +): + """Scenario: an admin config lists Codex models and launch uses its default or the + customer first-prompt-and-subagent selector. + + Expected: `ug configure` applies the config without the personal agent selector, and both + launch modes route the first real prompt: the TUI shows the Unity Gateway Smart Router + banner naming the selected model, the routed answer completes the file task, and the session + exits normally. """ session = live_session task = FileTask(session) use_managed_config_fixture(session, "codex_smart_routing") result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) assert "Select coding agents to configure:" not in result.stdout, result.stdout + if SMART_ROUTER_CONFIG_VERSION is not None: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION with AgentTerminal( - session, "codex", [str(session.binary), "codex"], "managed-smart-routing" + session, + "codex", + [str(session.binary), "codex"], + f"managed-smart-routing-{SMART_ROUTER_CONFIG_VERSION or 'managed-default'}", ) as tui: tui.boot() tui.submit(task.prompt) diff --git a/tests/integration/test_ug_smart_routing_hooks.py b/tests/integration/test_ug_smart_routing_hooks.py index fc4c92db9..1aec53ab9 100644 --- a/tests/integration/test_ug_smart_routing_hooks.py +++ b/tests/integration/test_ug_smart_routing_hooks.py @@ -111,7 +111,9 @@ def _run_calculation(tui, session, agent: str, expression: str, expected: str, * ) -def _toggle_with_skill(tui, session, agent: str, enabled: bool) -> None: +def _toggle_with_skill( + tui, session, agent: str, enabled: bool, orchestration_enabled: bool +) -> None: skill_root = session.home / SKILL_ROOTS[agent] ignored_skills = {".system"} if agent == "codex" else set() installed_skills = sorted( @@ -120,9 +122,7 @@ def _toggle_with_skill(tui, session, agent: str, enabled: bool) -> None: if path.is_dir() and path.name not in ignored_skills ) expected_skills = ( - ["smart-router", "smart-router-orchestrator"] - if session.env.get("ENABLE_SMART_ROUTER_ORCHESTRATOR") == "1" - else ["smart-router"] + ["smart-router", "smart-router-orchestrator"] if orchestration_enabled else ["smart-router"] ) assert installed_skills == expected_skills, installed_skills @@ -136,6 +136,7 @@ def _toggle_with_skill(tui, session, agent: str, enabled: bool) -> None: else { "ENABLE_SMART_ROUTING_V2": "0", "ENABLE_SMART_ROUTING_SUBAGENT_ONLY": "0", + "ENABLE_SMART_ROUTER_ORCHESTRATOR": "0", } ) assert json.loads(controls[0].read_text()) != expected @@ -171,9 +172,28 @@ def toggled(_screen): @pytest.mark.live @pytest.mark.claude -def test_smart_routing_claude_route_subagent_hook(live_session, workspace): - """Scenario: with subagent-only routing enabled, Claude Code fires PreToolUse for an - Agent spawn, piping the payload to ``ug claude-router-hook route-subagent``. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + None, + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], + ids=[ + "legacy-env", + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], +) +def test_smart_routing_claude_route_subagent_hook( + live_session, workspace, SMART_ROUTER_CONFIG_VERSION +): + """Scenario: legacy subagent routing or each supported selector makes Claude Code fire + PreToolUse for an Agent spawn, piping the payload to ``ug claude-router-hook route-subagent``. Expected: the hook allows the call against the real workspace router, drops the requested model in favor of a ``ucode-route-`` agent definition while preserving the @@ -181,7 +201,10 @@ def test_smart_routing_claude_route_subagent_hook(live_session, workspace): hook contract is asserted; no agent decides to spawn. """ session = live_session - session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" + if SMART_ROUTER_CONFIG_VERSION is None: + session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" + else: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION payload = { "session_id": "claude-route-subagent-hook", "tool_name": "Agent", @@ -225,9 +248,28 @@ def test_smart_routing_claude_route_subagent_hook(live_session, workspace): @pytest.mark.live @pytest.mark.codex -def test_smart_routing_codex_route_subagent_hook(live_session, workspace): - """Scenario: with subagent-only routing enabled, Codex fires PreToolUse for a - spawn_agent call, piping the payload to ``ug codex-router-hook route-subagent``. +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + None, + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], + ids=[ + "legacy-env", + "first_prompt_and_subagent_no_orch_v0", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], +) +def test_smart_routing_codex_route_subagent_hook( + live_session, workspace, SMART_ROUTER_CONFIG_VERSION +): + """Scenario: legacy subagent routing or each supported selector makes Codex fire PreToolUse + for a spawn_agent call, piping the payload to ``ug codex-router-hook route-subagent``. Expected: the hook allows the call against the real workspace router, rewrites the requested model to the bundled catalog slug of an offered model while preserving the @@ -235,7 +277,10 @@ def test_smart_routing_codex_route_subagent_hook(live_session, workspace): hook contract is asserted; no agent decides to spawn. """ session = live_session - session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" + if SMART_ROUTER_CONFIG_VERSION is None: + session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" + else: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION payload = { "session_id": "codex-route-subagent-hook", "tool_name": "spawn_agent", @@ -279,29 +324,46 @@ def test_smart_routing_codex_route_subagent_hook(live_session, workspace): @pytest.mark.claude @pytest.mark.managed_fixture @pytest.mark.parametrize( - "orchestration_enabled", [False, True], ids=["routing-only", "orchestration"] + "orchestration_enabled, SMART_ROUTER_CONFIG_VERSION", + [ + (False, None), + (True, None), + (False, "subagent_only_v0"), + (False, "subagent_only_v1"), + (True, "subagent_orch_v0"), + ], + ids=[ + "routing-only", + "orchestration", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], ) def test_smart_router_skill_toggles_claude_subagent_routing( - live_session, workspace, tmp_path, orchestration_enabled + live_session, workspace, tmp_path, orchestration_enabled, SMART_ROUTER_CONFIG_VERSION ): - """Scenario: launch Claude with subagent routing enabled and orchestration unset - or opted in through ENABLE_SMART_ROUTER_ORCHESTRATOR=1, spawn a child, invoke the - installed Smart Router skill to turn routing off, spawn another child, turn routing + """Scenario: launch Claude with the legacy flags or a subagent selector, spawn a child, invoke + the installed Smart Router skill to turn routing off, spawn another child, turn routing back on through the skill, and spawn a third child in the same real TUI session. - Expected: only Smart Router is installed by default; opting in also installs smart-router-orchestrator. - Each invocation records the CLI confirmation in the native transcript and changes the saved - routing controls, even with collapsed terminal output; all three uniquely tagged - calculations complete in native child sessions; only the first and third show the - subagent-routing banner and produce live gateway decisions correlated with those children. - No first-prompt routing wrapper starts. + Expected: routing-only cases install Smart Router; enabling orchestration also installs + Smart Router Orchestrator. Each invocation records the CLI + confirmation in the native transcript and changes the saved routing controls, even with + collapsed terminal output; all three uniquely tagged calculations complete in native child + sessions; only the first and third show the subagent-routing banner and produce live gateway + decisions correlated with those children. Claude's native task view reports no running + tasks before /exit is submitted. No first-prompt routing wrapper starts. """ session = live_session session.env["TMPDIR"] = str(tmp_path) - session.env["ENABLE_SMART_ROUTING_V2"] = "1" - session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" - if orchestration_enabled: - session.env["ENABLE_SMART_ROUTER_ORCHESTRATOR"] = "1" + if SMART_ROUTER_CONFIG_VERSION is None: + session.env["ENABLE_SMART_ROUTING_V2"] = "1" + session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" + if orchestration_enabled: + session.env["ENABLE_SMART_ROUTER_ORCHESTRATOR"] = "1" + else: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION use_managed_config_fixture(session, "claude_smart_routing") session.run( "configure", @@ -316,10 +378,23 @@ def test_smart_router_skill_toggles_claude_subagent_routing( ) as tui: tui.boot() _run_calculation(tui, session, "claude", "1+1", "2", routed=True) - _toggle_with_skill(tui, session, "claude", enabled=False) + _toggle_with_skill( + tui, + session, + "claude", + enabled=False, + orchestration_enabled=orchestration_enabled, + ) _run_calculation(tui, session, "claude", "1+2", "3", routed=False) - _toggle_with_skill(tui, session, "claude", enabled=True) + _toggle_with_skill( + tui, + session, + "claude", + enabled=True, + orchestration_enabled=orchestration_enabled, + ) _run_calculation(tui, session, "claude", "2+2", "4", routed=True) + tui.wait_for_background_tasks() tui.exit_normally() transcript = "".join(tui.output) assert SMART_ROUTING_BANNER not in transcript, transcript @@ -332,29 +407,45 @@ def test_smart_router_skill_toggles_claude_subagent_routing( @pytest.mark.codex @pytest.mark.managed_fixture @pytest.mark.parametrize( - "orchestration_enabled", [False, True], ids=["routing-only", "orchestration"] + "orchestration_enabled, SMART_ROUTER_CONFIG_VERSION", + [ + (False, None), + (True, None), + (False, "subagent_only_v0"), + (False, "subagent_only_v1"), + (True, "subagent_orch_v0"), + ], + ids=[ + "routing-only", + "orchestration", + "subagent_only_v0", + "subagent_only_v1", + "subagent_orch_v0", + ], ) def test_smart_router_skill_toggles_codex_subagent_routing( - live_session, workspace, tmp_path, orchestration_enabled + live_session, workspace, tmp_path, orchestration_enabled, SMART_ROUTER_CONFIG_VERSION ): - """Scenario: launch Codex with subagent routing enabled and orchestration unset - or opted in through ENABLE_SMART_ROUTER_ORCHESTRATOR=1, spawn a child, invoke the - installed Smart Router skill to turn routing off, spawn another child, turn routing + """Scenario: launch Codex with the legacy flags or a subagent selector, spawn a child, invoke + the installed Smart Router skill to turn routing off, spawn another child, turn routing back on through the skill, and spawn a third child in the same real TUI session. - Expected: only Smart Router is installed by default; opting in also installs smart-router-orchestrator. - Each invocation records the CLI confirmation in the native transcript and changes the saved - routing controls, even with collapsed terminal output; all three uniquely tagged - calculations complete in native child sessions; only the first and third show the - subagent-routing banner and produce live gateway decisions correlated with those children. - No first-prompt interposer starts. + Expected: routing-only cases install Smart Router; enabling orchestration also installs + Smart Router Orchestrator. Each invocation records the CLI + confirmation in the native transcript and changes the saved routing controls, even with + collapsed terminal output; all three uniquely tagged calculations complete in native child + sessions; only the first and third show the subagent-routing banner and produce live gateway + decisions correlated with those children. No first-prompt interposer starts. """ session = live_session session.env["TMPDIR"] = str(tmp_path) - session.env["ENABLE_SMART_ROUTING_V2"] = "1" - session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" - if orchestration_enabled: - session.env["ENABLE_SMART_ROUTER_ORCHESTRATOR"] = "1" + if SMART_ROUTER_CONFIG_VERSION is None: + session.env["ENABLE_SMART_ROUTING_V2"] = "1" + session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" + if orchestration_enabled: + session.env["ENABLE_SMART_ROUTER_ORCHESTRATOR"] = "1" + else: + session.env["SMART_ROUTER_CONFIG_VERSION"] = SMART_ROUTER_CONFIG_VERSION use_managed_config_fixture(session, "codex_smart_routing") session.run( "configure", @@ -369,9 +460,21 @@ def test_smart_router_skill_toggles_codex_subagent_routing( ) as tui: tui.boot() _run_calculation(tui, session, "codex", "1+1", "2", routed=True) - _toggle_with_skill(tui, session, "codex", enabled=False) + _toggle_with_skill( + tui, + session, + "codex", + enabled=False, + orchestration_enabled=orchestration_enabled, + ) _run_calculation(tui, session, "codex", "1+2", "3", routed=False) - _toggle_with_skill(tui, session, "codex", enabled=True) + _toggle_with_skill( + tui, + session, + "codex", + enabled=True, + orchestration_enabled=orchestration_enabled, + ) _run_calculation(tui, session, "codex", "2+2", "4", routed=True) tui.exit_normally() transcript = "".join(tui.output) diff --git a/tests/integration/utils/terminal.py b/tests/integration/utils/terminal.py index 85240dd59..299561298 100644 --- a/tests/integration/utils/terminal.py +++ b/tests/integration/utils/terminal.py @@ -19,6 +19,31 @@ from .evidence import agent_sessions, assert_no_terminal_api_error +def _claude_background_task_menu(text): + return re.search( + r"(?ms)^[ \t]*Background[ \t]*\n(.*?)" + r"^[ \t]*↑/↓ to select · Enter to view · Esc to close[ \t]*\s*\Z", + text, + ) + + +def _claude_background_tasks_complete(text): + if re.search(r"(?m)^\s*No tasks currently running\s*$", text): + return True + menu = _claude_background_task_menu(text) + if menu is None: + return False + rows = [line.strip() for line in menu[1].splitlines() if line.strip()] + if not rows: + return False + completed = re.fullmatch(r"Completed \(([1-9][0-9]*)\)", rows[0]) + return ( + completed is not None + and len(rows) - 1 == int(completed[1]) + and all(re.fullmatch(r"(?:❯\s*)?✔\s+.+\s+done\s+·\s+.+", row) for row in rows[1:]) + ) + + class TerminalScreen(pyte.Screen): def __init__(self, columns, lines, send): super().__init__(columns, lines) @@ -400,6 +425,24 @@ def completed(screen): ) task.assert_completed(self.session, self.agent) + def wait_for_background_tasks(self, timeout=180): + """Wait in Claude's native task view without stopping or detaching work.""" + assert self.agent == "claude", self.agent + self.submit("/tasks") + self.wait_for( + _claude_background_tasks_complete, + "Claude's task view reporting no running tasks", + timeout=timeout, + ) + self.send("\x1b", "close the completed background-task view") + self.wait_for( + lambda text: ( + "No tasks currently running" not in text + and _claude_background_task_menu(text) is None + ), + "the prompt after closing the background-task view", + ) + def exit_normally(self): self.submit("/exit") self.finish(timeout=30) diff --git a/tests/test_cli.py b/tests/test_cli.py index a9d9a859c..b95af40d1 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -1875,7 +1875,11 @@ def _isolated_bearer(self): else: os.environ["DATABRICKS_BEARER"] = original - def test_prints_only_the_token_to_stdout(self): + @pytest.mark.parametrize("selector", [None, "unsupported_v99"]) + def test_prints_only_the_token_to_stdout(self, monkeypatch, selector): + if selector is not None: + monkeypatch.setenv("SMART_ROUTER_CONFIG_VERSION", selector) + previous = dict(os.environ) with ( patch("ucode.cli.load_state", return_value={"workspace": "https://ws"}), patch("ucode.cli.get_databricks_token", return_value="tok-123") as fetch, @@ -1886,6 +1890,7 @@ def test_prints_only_the_token_to_stdout(self): # or the consuming agent will treat the noise as part of the token. assert result.stdout == "tok-123\n" fetch.assert_called_once_with("https://ws", None, force_refresh=False) + assert dict(os.environ) == previous def test_host_and_profile_override_state(self): with ( @@ -2844,6 +2849,17 @@ def test_skills_entry_absent_from_per_client_mcp_lines(self): class TestRevert: + def test_invalid_routing_config_does_not_block_revert(self, monkeypatch): + monkeypatch.setenv("SMART_ROUTER_CONFIG_VERSION", "unsupported_v99") + monkeypatch.setenv("ENABLE_SMART_ROUTING_V2", "1") + previous = dict(os.environ) + with patch("ucode.cli.revert", return_value=0) as revert: + result = runner.invoke(app, ["revert"]) + + assert result.exit_code == 0, result.output + revert.assert_called_once_with() + assert dict(os.environ) == previous + def test_reverts_mcp_configs_before_clearing_state(self): state = { **MINIMAL_STATE, @@ -2871,6 +2887,28 @@ def test_reverts_mcp_configs_before_clearing_state(self): assert "Claude Code MCP config: restored" in result.output +@pytest.mark.parametrize("args", [[], ["claude"], ["codex"]]) +def test_unknown_routing_config_does_not_block_launch(monkeypatch, args): + monkeypatch.setenv("SMART_ROUTER_CONFIG_VERSION", "unsupported_v99") + previous = dict(os.environ) + with ( + patch("ucode.cli._launch_tool") as launch, + patch("ucode.cli._launch_managed_default") as launch_default, + patch("ucode.cli.configure_shared_state") as configure, + ): + result = runner.invoke(app, args) + + assert result.exit_code == 0, result.output + if args: + launch.assert_called_once() + launch_default.assert_not_called() + else: + launch.assert_not_called() + launch_default.assert_called_once() + configure.assert_not_called() + assert dict(os.environ) == previous + + class TestDoctorCommand: def test_invokes_doctor(self): with patch("ucode.doctor.doctor", return_value=0) as mock_doctor: diff --git a/tests/test_cuj_evidence.py b/tests/test_cuj_evidence.py index fa50ef2ef..40f84ba74 100644 --- a/tests/test_cuj_evidence.py +++ b/tests/test_cuj_evidence.py @@ -6,7 +6,7 @@ import pytest -from tests.e2e_cuj.helpers.constants import CLAUDE, CODEX +from tests.e2e_cuj.helpers.constants import CLAUDE, CODEX, INFERENCE_PATHS from tests.e2e_cuj.helpers.evidence import ( BaseCujHelper, ClaudeCujHelper, @@ -16,6 +16,8 @@ completed_turn, get_cuj_helper, ) +from tests.e2e_cuj.helpers.tui_request_recorder import RecordedRequest +from tests.e2e_cuj.test_cuj4_smart_routing import _task_inference_request from tests.integration.utils.evidence import FileTask, read_jsonl @@ -101,6 +103,24 @@ def test_cuj_evidence_completed_native_turn(tmp_path, agent): assert result["selected_model"] == model +def test_cuj_evidence_claude_real_user_message_ends_parent_turn(): + task = SimpleNamespace( + prompt="Delegate reading the file to one subagent.", value="hidden-value" + ) + prompt, answer = records(CLAUDE, task, "system.ai.claude-haiku-4-5") + user = { + "type": "user", + "sessionId": "session", + "origin": {"kind": "human"}, + "message": {"content": "A different task."}, + } + assert completed_turn(CLAUDE, [prompt, user, answer], task) is None + user["message"]["content"] = ( + f"{task.value}" + ) + assert completed_turn(CLAUDE, [prompt, user, answer], task) is None + + @pytest.mark.parametrize("agent", [CLAUDE, CODEX]) @pytest.mark.parametrize( "failure", @@ -149,6 +169,119 @@ def test_cuj_evidence_codex_rejects_wrong_turn(tmp_path): assert completed_turn(CODEX, wrong_turn, task) is None +def _tagged_codex_notification(row): + notification = copy.deepcopy(row) + notification["payload"]["content"] = [{"type": "input_text", "text": ""}] + notification["internal_chat_message_metadata_passthrough"] = { + "content_item_kinds": ["multi_agent.subagent_notification"] + } + return notification + + +def _codex_parent_delegation_records(task, model): + rows = records(CODEX, SimpleNamespace(prompt="initial", value="initial"), model) + target = records(CODEX, task, model)[1:] + for row in target: + if row["type"] != "response_item": + row["payload"]["turn_id"] = "delegate" + rows.extend( + [ + {"type": "event_msg", "payload": {"type": "user_message", "message": task.prompt}}, + *target[:3], + _tagged_codex_notification(target[2]), + target[3], + ] + ) + return rows + + +def test_cuj_evidence_codex_delegated_parent_turn(tmp_path): + task = FileTask(SimpleNamespace(cwd=tmp_path)) + task.prompt = task.delegate_prompt + rows = _codex_parent_delegation_records(task, "gpt-6-sol") + later = records(CODEX, SimpleNamespace(prompt="later", value="later"), "gpt-6-sol")[1:] + for row in later: + if row["type"] != "response_item": + row["payload"]["turn_id"] = "later" + rows.extend(later) + turn = completed_turn(CODEX, rows, task) + assert turn is not None + assert (turn.turn_id, turn.answer) == ("delegate", task.value) + + +@pytest.mark.parametrize("case", ["aborted", "later", "notification", "duplicate"]) +def test_cuj_evidence_codex_rejects_follow_up_false_positives(tmp_path, case): + task = FileTask(SimpleNamespace(cwd=tmp_path)) + rows = records(CODEX, task, "gpt-6-sol") + if case == "aborted": + rows[-1]["payload"]["type"] = "turn_aborted" + elif case == "later": + rows.pop() + rows.append({"type": "event_msg", "payload": {"type": "task_started", "turn_id": "later"}}) + elif case == "notification": + rows[3] = _tagged_codex_notification(rows[3]) + else: + rows.insert(-1, copy.deepcopy(rows[3])) + if case == "duplicate": + with pytest.raises(AssertionError, match="Prompt was submitted more than once"): + completed_turn(CODEX, rows, task) + else: + assert completed_turn(CODEX, rows, task) is None + + +def _inference_request(agent, sequence, prompt, tools): + field = "messages" if agent == CLAUDE else "input" + content = prompt if agent == CLAUDE else [{"type": "input_text", "text": prompt}] + return SimpleNamespace( + method="POST", + path=INFERENCE_PATHS[agent], + sequence=sequence, + payload={field: [{"role": "user", "content": content}], "tools": tools}, + ) + + +@pytest.mark.parametrize("agent", [CLAUDE, CODEX]) +@pytest.mark.parametrize( + ("method", "path", "sequence"), + [("GET", None, 2), ("POST", "/models", 2), ("POST", None, 1)], +) +def test_cuj_task_inference_skips_unrelated_empty_bodies(agent, method, path, sequence): + unrelated = RecordedRequest(sequence, method, path or INFERENCE_PATHS[agent], {}, b"") + task = _inference_request(agent, 3, "task prompt", [{"name": "Read"}]) + assert _task_inference_request([unrelated, task], agent, "task prompt", after=1) is task + + +def test_cuj_task_inference_selection_uses_exact_tool_prompt(): + prompt = "Read input-file.txt using a tool. Reply with only its contents." + claude = [ + _inference_request(CLAUDE, 1, f"Name this session: {prompt}", []), + _inference_request(CLAUDE, 2, prompt, []), + _inference_request(CLAUDE, 3, prompt, [{"name": "Read"}]), + ] + assert _task_inference_request(claude, CLAUDE, prompt) is claude[2] + child_prompt = "Read input-file.txt using a tool and return its contents." + parent = _inference_request( + CODEX, 4, "Delegate this task to one subagent.", [{"name": "spawn_agent"}] + ) + child = _inference_request(CODEX, 5, child_prompt, [{"name": "read_file"}]) + assert _task_inference_request([parent, child], CODEX, child_prompt, after=3) is child + with pytest.raises(AssertionError, match="No tool-capable"): + _task_inference_request([parent, child], CODEX, child_prompt, after=5) + + +@pytest.mark.parametrize("agent", [CLAUDE, CODEX]) +@pytest.mark.parametrize("after", [0, 1]) +@pytest.mark.parametrize("suffix", ["\n", "\n\n", " ", "\nDifferent task."]) +def test_cuj_task_inference_only_allows_claude_child_transport_newline(agent, after, suffix): + prompt = "Read input-file.txt using a tool and return its contents." + request = _inference_request(agent, 2, prompt + suffix, [{"name": "Read"}]) + if agent == CLAUDE and after and suffix == "\n": + assert _task_inference_request([request], agent, prompt, after=after) is request + else: + with pytest.raises(AssertionError, match="No tool-capable"): + _task_inference_request([request], agent, prompt, after=after) + + def test_cuj_evidence_ignores_existing_session(tmp_path): task = FileTask(SimpleNamespace(cwd=tmp_path)) first = SessionEvidence(tmp_path, CLAUDE) diff --git a/tests/test_e2e_cuj_helpers.py b/tests/test_e2e_cuj_helpers.py index edecb09b9..cf32874a1 100644 --- a/tests/test_e2e_cuj_helpers.py +++ b/tests/test_e2e_cuj_helpers.py @@ -20,7 +20,11 @@ INFERENCE_PATHS, CodingAgent, ) -from tests.e2e_cuj.helpers.evidence import assert_models, claude_file_task +from tests.e2e_cuj.helpers.evidence import ( + assert_models, + claude_file_task, + served_inference_request, +) from tests.e2e_cuj.helpers.poll import poll from tests.e2e_cuj.helpers.session import ( MACHINE_WIDE_LEAK, @@ -31,6 +35,7 @@ from tests.e2e_cuj.helpers.terminal import Terminal from tests.e2e_cuj.helpers.workspace import Workspace from tests.e2e_cuj.test_cuj3_models import _assert_inference_evidence, _catalog_display_names +from tests.e2e_cuj.test_cuj4_smart_routing import _task_inference_request def _client(headers): @@ -164,6 +169,133 @@ def approve(screen): assert "allowed" in tui.visible +@requires_pty +def test_wait_until_acknowledges_ready_billing_notices_and_observes_dismissal(tmp_path): + """A rendered notice can precede its input handler and appear again for a child.""" + done = tmp_path / "done" + script = tmp_path / "ug" + script.write_text( + f"""#!{sys.executable} +import sys +import termios +import time +import tty +from pathlib import Path + +tty.setraw(sys.stdin.fileno()) +for _ in range(2): + print("\\x1b[2J\\x1b[HWe're changing auto mode to no longer charge for classifier requests", flush=True) + print("Nothing breaks: auto mode keeps working", flush=True) + print("Enter to continue · Esc to cancel", flush=True) + time.sleep(0.2) + termios.tcflush(sys.stdin.fileno(), termios.TCIFLUSH) + assert sys.stdin.read(1) == "\\r" + print("\\x1b[2J\\x1b[HWorking", flush=True) + time.sleep(0.8) +Path({str(done)!r}).touch() +time.sleep(30) +""" + ) + script.chmod(0o755) + session = UserSession(tmp_path, script, tmp_path / "artifacts", "token") + with Terminal(session, "billing-notices", [], agent=CLAUDE) as tui: + tui.wait_until(done.exists, "completed work after both billing notices", timeout=10) + acknowledgements = [ + action + for action in tui.actions + if action["reason"] == "acknowledge auto-mode classifier billing notice" + ] + assert len(acknowledgements) == 2 + assert all(action["keys"] == "\r" for action in acknowledgements) + assert "Enter to continue" not in tui.visible + + +@requires_pty +@pytest.mark.parametrize( + "view", + ["empty", "completed", "resumed", "running", "mixed", "scheduled", "partial", "transcript"], +) +def test_claude_waits_for_background_task_completion_before_exit(tmp_path, view): + """An offline terminal fixture requires /tasks, completion, Escape, then /exit.""" + views = { + "empty": "No tasks currently running", + "completed": ( + "Background\n\n Completed (2)\n" + "❯ ✔ ug_subagent_two done · Haiku 4.5\n" + " ✔ ug_subagent_one done · Sonnet 5\n\n" + "↑/↓ to select · Enter to view · Esc to close" + ), + "running": ( + "Background\n\n Running (1)\n❯ ug_subagent_one running · Sonnet 5\n\n" + "↑/↓ to select · Enter to view · Esc to close" + ), + "mixed": ( + "Background\n\n Running (1)\n❯ ug_subagent_one running · Sonnet 5\n" + " Completed (1)\n ✔ ug_subagent_two done · Haiku 4.5\n\n" + "↑/↓ to select · Enter to view · Esc to close" + ), + "scheduled": ( + "Background\n\n Scheduled (1)\n❯ scheduled task · Runs once in 1m\n" + " Completed (1)\n ✔ ug_subagent_two done · Haiku 4.5\n\n" + "↑/↓ to select · Enter to view · Esc to close" + ), + "partial": ( + "Background\n\n Completed (2)\n❯ ✔ ug_subagent_one done · Sonnet 5\n\n" + "↑/↓ to select · Enter to view · Esc to close" + ), + "transcript": ("Background\n\n Completed (1)\n✔ ug_subagent_one done · Sonnet 5\n\n❯"), + } + views["resumed"] = views["mixed"] + complete = view in ("empty", "completed", "resumed") + script = tmp_path / "ug" + script.write_text( + f"""#!{sys.executable} +import sys +import select +import termios +import time +import tty + +print("❯", flush=True) +assert sys.stdin.readline().strip() == "/tasks" +print("Background tasks: scheduled task · Runs once in 1m", flush=True) +time.sleep(0.8) +previous = termios.tcgetattr(sys.stdin.fileno()) +tty.setraw(sys.stdin.fileno()) +print("\\x1b[2J\\x1b[H" + {views[view]!r}.replace("\\n", "\\r\\n"), flush=True) +if {view == "resumed"!r}: + time.sleep(0.8) + assert not select.select([sys.stdin], [], [], 0)[0], "Exited with a resumed child running" + print("\\x1b[2J\\x1b[H" + {views["completed"]!r}.replace("\\n", "\\r\\n"), flush=True) +assert sys.stdin.read(1) == "\\x1b" +termios.tcsetattr(sys.stdin.fileno(), termios.TCSANOW, previous) +print("\\x1b[2J\\x1b[H❯", flush=True) +assert sys.stdin.readline().strip() == "/exit" +""" + ) + script.chmod(0o755) + session = UserSession(tmp_path, script, tmp_path / "artifacts", "token") + with Terminal(session, "waits-before-exit", [], agent=CLAUDE) as tui: + if complete: + tui.wait_for_background_tasks(timeout=5) + tui.exit_normally() + assert tui.child.exitstatus == 0 + close = next( + action + for action in tui.actions + if action["reason"] == "close the completed background-task view" + ) + assert views[view].splitlines()[0] in close["screen_before"] + if view == "resumed": + assert "Running (1)" not in close["screen_before"] + assert "Completed (2)" in close["screen_before"] + else: + with pytest.raises(AssertionError, match="task view reporting no running tasks"): + tui.wait_for_background_tasks(timeout=2) + assert not tui.ended + assert not any("/exit" in action["keys"] for action in tui.actions) + + def test_mcp_list_poll_retries_a_failed_probe_until_rows_match(monkeypatch): monkeypatch.setattr(poll_module.time, "sleep", lambda seconds: None) healthy = "\n".join( @@ -251,29 +383,40 @@ def test_catalog_display_names_rejects_repeated_pages(pages, message): _catalog_display_names(workspace, CLAUDE, "ug_e2e.models") -@pytest.fixture -def thinking_display_exchange(): +@pytest.fixture(params=["adaptive", "enabled"]) +def thinking_display_exchange(request): model = "ug_e2e.models.claude_sonnet" task = SimpleNamespace(prompt="Read the task file") + thinking = {"type": request.param, "display": "updates"} + if request.param == "enabled": + thinking["budget_tokens"] = 31999 payload = { "model": model, "messages": [{"role": "user", "content": task.prompt}], - "thinking": {"type": "adaptive", "display": "updates"}, + "tools": [{"name": "Read"}], + "thinking": thinking, "output_config": {"effort": "high"}, } requests = [ - SimpleNamespace(method="POST", path=INFERENCE_PATHS[CLAUDE], payload=payload), + SimpleNamespace(sequence=1, method="POST", path=INFERENCE_PATHS[CLAUDE], payload=payload), SimpleNamespace( + sequence=2, method="POST", path=INFERENCE_PATHS[CLAUDE], - payload={**payload, "thinking": {"type": "adaptive"}}, + payload={ + **payload, + "thinking": {key: value for key, value in thinking.items() if key != "display"}, + }, ), ] error = json.dumps( { "error_code": "BAD_REQUEST", "message": json.dumps( - {"message": "thinking.adaptive.display: Input should be 'summarized', 'omitted'"} + { + "message": f"thinking.{request.param}.display: " + "Input should be 'summarized', 'omitted'" + } ), } ).encode() @@ -288,15 +431,20 @@ def thinking_display_exchange(): return recorder, requests, responses, task, model +@pytest.mark.parametrize("contract", ["catalog", "routing"]) @pytest.mark.parametrize("compressed", [False, True]) -def test_catalog_inference_accepts_verified_thinking_display_recovery( - thinking_display_exchange, compressed +def test_cuj_inference_accepts_verified_thinking_display_recovery( + thinking_display_exchange, compressed, contract ): - recorder, _, responses, task, model = thinking_display_exchange + recorder, requests, responses, task, model = thinking_display_exchange if compressed: responses[0].body = gzip.compress(responses[0].body) responses[0].headers = {"content-encoding": "gzip"} - _assert_inference_evidence(recorder, 0, CLAUDE, task, model) + if contract == "catalog": + _assert_inference_evidence(recorder, 0, CLAUDE, task, model) + else: + inference = _task_inference_request(requests, CLAUDE, task.prompt) + assert served_inference_request(recorder, requests, inference, CLAUDE) is requests[1] @pytest.mark.parametrize( @@ -310,17 +458,19 @@ def test_catalog_inference_accepts_verified_thinking_display_recovery( "empty_retry", "changed_model", "changed_effort", + "changed_budget", "changed_prompt", "display_retained", "wrong_display", "codex", ], ) -def test_catalog_inference_rejects_unverified_recovery(thinking_display_exchange, failure): +@pytest.mark.parametrize("contract", ["catalog", "routing"]) +def test_cuj_inference_rejects_unverified_recovery(thinking_display_exchange, failure, contract): recorder, requests, responses, task, model = thinking_display_exchange agent = CLAUDE if failure == "unrelated_400": - responses[0].body = responses[0].body.replace(b"thinking.adaptive.display", b"other.field") + responses[0].body = responses[0].body.replace(b".display", b".other_field") elif failure == "malformed_error": responses[0].body = b"not JSON" elif failure == "server_error": @@ -335,6 +485,8 @@ def test_catalog_inference_rejects_unverified_recovery(thinking_display_exchange requests[1].payload["model"] = "ug_e2e.models.claude_haiku" elif failure == "changed_effort": requests[1].payload["output_config"] = {} + elif failure == "changed_budget": + requests[1].payload["thinking"]["budget_tokens"] = 1000 elif failure == "changed_prompt": requests[1].payload["messages"] = [{"role": "user", "content": "A different task"}] elif failure == "display_retained": @@ -347,7 +499,148 @@ def test_catalog_inference_rejects_unverified_recovery(thinking_display_exchange request.path = INFERENCE_PATHS[CODEX] request.payload["input"] = task.prompt with pytest.raises(AssertionError): - _assert_inference_evidence(recorder, 0, agent, task, model) + if contract == "catalog": + _assert_inference_evidence(recorder, 0, agent, task, model) + else: + inference = _task_inference_request(requests, agent, task.prompt) + served_inference_request(recorder, requests, inference, agent) + + +@pytest.mark.parametrize("agent", [CLAUDE, CODEX]) +def test_served_inference_request_keeps_successful_first_attempt(thinking_display_exchange, agent): + recorder, requests, responses, _, _ = thinking_display_exchange + responses[0].status_code = 200 + responses[0].body = b"successful stream" + assert served_inference_request(recorder, requests, requests[0], agent) is requests[0] + + +@pytest.mark.parametrize("context", ["system", "metadata"]) +@pytest.mark.parametrize("changed_model", [False, True]) +def test_served_inference_matches_retry_despite_interleaved_traffic( + thinking_display_exchange, context, changed_model +): + recorder, requests, responses, _, _ = thinking_display_exchange + rejected, retry = requests + unrelated = SimpleNamespace( + sequence=2, + method=rejected.method, + path=rejected.path, + payload={**retry.payload, context: "parent context"}, + ) + requests.insert(1, unrelated) + responses.insert(1, SimpleNamespace(status_code=200, headers={}, body=b"parent stream")) + if changed_model: + retry.payload["model"] = "wrong model" + with pytest.raises(AssertionError, match="changed more than the rejected field"): + served_inference_request(recorder, requests, rejected, CLAUDE) + else: + assert served_inference_request(recorder, requests, rejected, CLAUDE) is retry + + +@pytest.fixture +def safeguards_exchange(thinking_display_exchange): + recorder, requests, responses, task, model = thinking_display_exchange + requests[0].payload["safeguards"] = {"enabled": True} + requests[1].payload["safeguards"] = {"enabled": True} + responses[1].status_code = 400 + responses[1].body = json.dumps( + { + "error_code": "BAD_REQUEST", + "message": json.dumps({"message": "safeguards: Extra inputs are not permitted"}), + } + ).encode() + requests.append( + SimpleNamespace( + sequence=3, + method="POST", + path=INFERENCE_PATHS[CLAUDE], + payload={ + key: value for key, value in requests[1].payload.items() if key != "safeguards" + }, + ) + ) + responses.append(SimpleNamespace(status_code=200, headers={}, body=b"successful stream")) + return recorder, requests, responses, task, model + + +@pytest.mark.parametrize("contract", ["catalog", "routing"]) +@pytest.mark.parametrize("compressed", [False, True]) +def test_cuj_inference_accepts_chained_native_compatibility_recovery( + safeguards_exchange, contract, compressed +): + recorder, requests, responses, task, model = safeguards_exchange + if compressed: + for response in responses[:2]: + response.body = gzip.compress(response.body) + response.headers = {"content-encoding": "gzip"} + if contract == "catalog": + _assert_inference_evidence(recorder, 0, CLAUDE, task, model) + else: + inference = _task_inference_request(requests, CLAUDE, task.prompt) + assert served_inference_request(recorder, requests, inference, CLAUDE) is requests[2] + + +def test_served_inference_accepts_safeguards_recovery_without_display_rejection( + safeguards_exchange, +): + recorder, requests, _, _, _ = safeguards_exchange + assert served_inference_request(recorder, requests, requests[1], CLAUDE) is requests[2] + + +@pytest.mark.parametrize("contract", ["catalog", "routing"]) +@pytest.mark.parametrize( + "failure", + [ + "missing_retry", + "failed_retry", + "empty_retry", + "changed_model", + "changed_effort", + "changed_budget", + "changed_prompt", + "safeguards_retained", + "unrelated_error", + "missing_safeguards", + "codex", + ], +) +def test_cuj_inference_rejects_unverified_safeguards_recovery( + safeguards_exchange, contract, failure +): + recorder, requests, responses, task, model = safeguards_exchange + agent = CLAUDE + if failure == "missing_retry": + requests.pop() + elif failure == "failed_retry": + responses[2].status_code = 400 + elif failure == "empty_retry": + responses[2].body = b"" + elif failure == "changed_model": + requests[2].payload["model"] = "different-model" + elif failure == "changed_effort": + requests[2].payload["output_config"] = {} + elif failure == "changed_budget": + requests[2].payload["thinking"] = {"type": "enabled", "budget_tokens": 1000} + elif failure == "changed_prompt": + requests[2].payload["messages"] = [{"role": "user", "content": "different task"}] + elif failure == "safeguards_retained": + requests[2].payload["safeguards"] = requests[1].payload["safeguards"] + elif failure == "unrelated_error": + responses[1].body = responses[1].body.replace(b"safeguards:", b"unrelated:") + elif failure == "missing_safeguards": + for request in requests: + request.payload.pop("safeguards", None) + elif failure == "codex": + agent = CODEX + for request in requests: + request.path = INFERENCE_PATHS[CODEX] + request.payload["input"] = task.prompt + with pytest.raises(AssertionError): + if contract == "catalog": + _assert_inference_evidence(recorder, 0, agent, task, model) + else: + inference = _task_inference_request(requests, agent, task.prompt) + served_inference_request(recorder, requests, inference, agent) def test_assert_models_maps_native_aliases(): diff --git a/tests/test_integration_evidence.py b/tests/test_integration_evidence.py index cf04efc76..e0c22846d 100644 --- a/tests/test_integration_evidence.py +++ b/tests/test_integration_evidence.py @@ -7,6 +7,7 @@ from tests.integration.utils import evidence from tests.integration.utils.agents import claude, codex from tests.integration.utils.evidence import ( + FileTask, SubagentCalculation, assert_no_terminal_api_error, assistant_answer_contains, @@ -103,6 +104,19 @@ def test_tagged_calculation_requires_the_native_child_answer(tmp_path, agent): assert "1+1" in task.prompt +@pytest.mark.parametrize("agent", ["claude", "codex"]) +@pytest.mark.parametrize("child", [False, True]) +def test_file_task_child_completion_is_independent_of_parent_answer(tmp_path, agent, child): + session = _Session(tmp_path) + session.cwd = tmp_path + task = FileTask(session) + assert task.value not in task.delegate_prompt + assert not task.completed(session, agent, child=True) + _write_answer(tmp_path, agent, child=child, value=task.value) + assert task.completed(session, agent, child=True) == child + assert task.completed(session, agent) != child + + def test_codex_model_identity_uses_only_the_completed_answer_turn(): records = [ { diff --git a/tests/test_smart_router.py b/tests/test_smart_router.py index 81613c882..504cf7991 100644 --- a/tests/test_smart_router.py +++ b/tests/test_smart_router.py @@ -70,7 +70,15 @@ def test_skill_toggles_with_launch_installation_despite_shadowed_path(tmp_path, ) assert result.returncode == 0, result.stdout + result.stderr assert f"Smart Router is {'off' if action == 'disable' else 'on'}" in result.stdout - expected = dict.fromkeys(v2.SMART_ROUTING_ENV_KEYS, "0") if action == "disable" else {} + expected = ( + { + v2.ENABLE_SMART_ROUTING_ENV_VAR: "0", + v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + v2.ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + } + if action == "disable" + else {} + ) assert json.loads(session_path.read_text()) == expected diff --git a/tests/test_smart_routing_config.py b/tests/test_smart_routing_config.py new file mode 100644 index 000000000..c797902a0 --- /dev/null +++ b/tests/test_smart_routing_config.py @@ -0,0 +1,142 @@ +"""Component coverage for supported versioned smart-routing configuration.""" + +from __future__ import annotations + +import pytest + +from ucode.constants import ( + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, + ENABLE_SMART_ROUTING_ENV_VAR, + ENABLE_SUBAGENT_ROUTING_ENV_VAR, + SMART_ROUTER_CONFIG_VERSION_ENV_VAR, +) +from ucode.smart_routing import config, orchestrator, v2 + +_EXPECTED_PRESETS = { + config.FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0: { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + }, + config.SUBAGENT_ONLY_V0: { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + }, + config.SUBAGENT_ONLY_V1: { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "0", + }, + config.SUBAGENT_ORCH_V0: { + ENABLE_SMART_ROUTING_ENV_VAR: "0", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + }, + config.SUBAGENT_ORCH_V1: { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + }, +} + + +@pytest.mark.parametrize( + "selector", [None, "", " ", "unsupported_version", " unsupported_version "] +) +def test_absent_or_unknown_selector_is_a_no_op(selector): + applied = { + ENABLE_SMART_ROUTING_ENV_VAR: "1", + ENABLE_SUBAGENT_ROUTING_ENV_VAR: "0", + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR: "1", + "INHERITED_SETTING": "preserved", + } + expected = applied.copy() + if selector is not None: + applied[SMART_ROUTER_CONFIG_VERSION_ENV_VAR] = selector + original = applied.copy() + + assert config.resolve_environment(applied) == expected + assert applied == original + assert config.apply_config(applied) == {} + assert applied == original + + +@pytest.mark.parametrize("ENABLE_SMART_ROUTING_V2", [None, "0", "1"]) +@pytest.mark.parametrize("ENABLE_SMART_ROUTING_SUBAGENT_ONLY", [None, "0", "1"]) +@pytest.mark.parametrize("ENABLE_SMART_ROUTER_ORCHESTRATOR", [None, "0", "1"]) +@pytest.mark.parametrize( + "SMART_ROUTER_CONFIG_VERSION", + [ + None, + config.FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0, + config.SUBAGENT_ONLY_V0, + config.SUBAGENT_ONLY_V1, + config.SUBAGENT_ORCH_V0, + config.SUBAGENT_ORCH_V1, + ], +) +def test_smart_routing_config_cartesian_grid( + ENABLE_SMART_ROUTING_V2, + ENABLE_SMART_ROUTING_SUBAGENT_ONLY, + ENABLE_SMART_ROUTER_ORCHESTRATOR, + SMART_ROUTER_CONFIG_VERSION, +): + source = { + environment_key: value + for environment_key, value in ( + (ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SMART_ROUTING_V2), + (ENABLE_SUBAGENT_ROUTING_ENV_VAR, ENABLE_SMART_ROUTING_SUBAGENT_ONLY), + (ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, ENABLE_SMART_ROUTER_ORCHESTRATOR), + ) + if value is not None + } + source["UNRELATED_SETTING"] = "preserved" + if SMART_ROUTER_CONFIG_VERSION is not None: + source[SMART_ROUTER_CONFIG_VERSION_ENV_VAR] = SMART_ROUTER_CONFIG_VERSION + original = source.copy() + + expected = original.copy() + expected.pop(SMART_ROUTER_CONFIG_VERSION_ENV_VAR, None) + if SMART_ROUTER_CONFIG_VERSION is not None: + expected.update(_EXPECTED_PRESETS[SMART_ROUTER_CONFIG_VERSION]) + + resolved = config.resolve_environment(source) + + assert resolved == expected + assert source == original + + applied = source.copy() + previous = config.apply_config(applied) + + assert applied == expected + v2.restore_smart_routing_env(previous, applied) + assert applied == original + + +@pytest.mark.parametrize( + ("version", "expected_routing"), + [ + (config.SUBAGENT_ONLY_V1, (True, False, False)), + (config.SUBAGENT_ORCH_V1, (True, False, True)), + (config.FIRST_PROMPT_AND_SUBAGENT_NO_ORCH_V0, (True, True, False)), + ], +) +def test_legacy_preset_equivalence(version, expected_routing): + legacy_env = _EXPECTED_PRESETS[version].copy() + legacy_env["SMART_ROUTER_NAME"] = "m2-r315-quality-20260929" + preset_env = { + SMART_ROUTER_CONFIG_VERSION_ENV_VAR: version, + "SMART_ROUTER_NAME": legacy_env["SMART_ROUTER_NAME"], + } + resolved = config.resolve_environment(preset_env) + applied = preset_env.copy() + config.apply_config(applied) + assert applied == resolved + for environment in (legacy_env, preset_env, resolved, applied): + assert environment["SMART_ROUTER_NAME"] == legacy_env["SMART_ROUTER_NAME"] + assert ( + v2.smart_routing_enabled(environment), + v2.first_prompt_routing_enabled(environment), + orchestrator.feature_enabled(environment), + ) == expected_routing