From 982e586df506e99cb479d4341a5e48d50e539e10 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 14:55:09 +0000 Subject: [PATCH 01/25] Integrate orchestrator skill with smart routing --- AGENTS.md | 4 + README.md | 26 + justfile | 8 +- skills/orchestrate/scripts/configure.py | 438 ++++++++ skills/smart-router/SKILL.md | 13 +- src/ucode/agents/claude.py | 12 +- src/ucode/agents/codex.py | 9 +- src/ucode/cli.py | 14 +- src/ucode/codex_config.py | 121 +++ src/ucode/skills.py | 1 + src/ucode/smart_routing/orchestrator.py | 164 +++ src/ucode/smart_routing/session_env.py | 18 +- src/ucode/smart_routing/v2.py | 85 +- tests/README.md | 21 + tests/conftest.py | 8 + tests/integration/README.md | 18 + .../test_ug_smart_routing_hooks.py | 61 +- tests/integration/utils/evidence.py | 32 + tests/test_agent_claude.py | 5 +- tests/test_agent_codex.py | 66 ++ tests/test_claude_smart_routing_v2.py | 5 +- tests/test_claude_windows_smart_routing.py | 9 +- tests/test_cli.py | 28 + tests/test_codex_config.py | 154 +++ tests/test_codex_smart_routing_v2.py | 159 ++- tests/test_integration_evidence.py | 74 ++ tests/test_managed_files.py | 4 +- tests/test_orchestrator.py | 233 +++++ tests/test_orchestrator_config.py | 954 ++++++++++++++++++ tests/test_orchestrator_legacy_plugins.py | 194 ++++ tests/test_smart_router.py | 4 +- 31 files changed, 2853 insertions(+), 89 deletions(-) create mode 100644 skills/orchestrate/scripts/configure.py create mode 100644 src/ucode/smart_routing/orchestrator.py create mode 100644 tests/test_orchestrator.py create mode 100644 tests/test_orchestrator_config.py create mode 100644 tests/test_orchestrator_legacy_plugins.py diff --git a/AGENTS.md b/AGENTS.md index 804b5ce6f..93ad62b4d 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -70,6 +70,8 @@ Fields live in `~/.claude/ucode-settings.json` and the OS-managed settings file | Tracing | Ignore | Create/replace | The seven `CLAUDE_CODE_*`/`OTEL_*` trace keys and `otelHeadersHelper`; only when the config enables tracing | | `managedMcpServers` | Ignore | Merge | Add/update the config's MCP server entries; other entries left alone | | Smart-routing hooks | Merge | Merge | `PreToolUse`, `SessionStart`, `SubagentStart`; only `ug`'s own marked handlers, other hooks left alone | +| Orchestrator hooks | Merge | Merge | Launch-only `UserPromptSubmit` and compact `SessionStart` handlers for smart-routed sessions; read the same session controls as routing | +| `enabledPlugins["model-orchestrator@…"]` | Merge | Merge | Launch-only `false` overrides for installed legacy plugins, even when smart routing is off; saved settings and other plugins left alone | @@ -86,5 +88,7 @@ Fields live in `~/.codex/ucode.config.toml` and `/etc/codex/managed_config.toml` | `http_headers` | Merge | Merge | In `[model_providers.Databricks]`; merge `ug`'s routing headers by name, admin headers added under managed config | | `model_catalog_json` | Create/replace | Create/replace | In `~/.codex/config.toml`; `ug`'s own catalog reference, for a static model list | | `mcp_servers` | Ignore | Merge | Managed file; add/update the config's MCP server entries, other entries left alone | +| Smart-routing and orchestrator hooks | Merge | Merge | Launch-only hooks preserve native effective handlers, including trusted project hooks; only `PreToolUse`, `UserPromptSubmit`, and `SessionStart` are overridden; `features.hooks` is enabled for a smart-routed launch | +| `plugins."model-orchestrator@…".enabled` | Merge | Merge | Launch-only `false` overrides for legacy registrations, including Isaac-synced and project plugins, even when smart routing is off; saved settings and other plugins left alone | diff --git a/README.md b/README.md index e19175612..453e5fae3 100644 --- a/README.md +++ b/README.md @@ -246,6 +246,32 @@ is enabled, ug warns and falls back to subagent routing because the first-prompt wrapper requires a Unix terminal. The generated shell hooks expect Git Bash; PowerShell-only setups are not covered. +### Automatic orchestration + +Smart-routed Claude and Codex sessions install the bundled `orchestrate` and +`smart-router` skills. The orchestrator assigns bounded work to explorer, +researcher, worker, tester, and reviewer roles while the root plans, integrates, +and verifies results. Easy tasks and explicit requests not to delegate stay in +the root. + +Orchestration follows the existing smart-routing launch eligibility and session +controls; it has no separate rollout flag. Turning Smart Router off stops new +automatic delegation, including fallback to default role models. Turning it on +restores orchestration. Explicit user requests for subagents still use normal +harness behavior while routing is off. Stored skill files do not activate it in +later non-routed sessions. The existing Isaac pilot gate and UG launch exclusions +are unchanged. + +UG automatically suppresses installed standalone `model-orchestrator` plugins, +including Isaac-synced Codex registrations, for each Claude or Codex launch. +This also applies when smart routing is off, so the old hooks cannot activate +orchestration independently. Codex project registrations are included. Routed Codex +launches use its native configuration resolver to preserve trusted project hooks; +untrusted project hooks stay disabled. Saved plugin settings, unrelated plugins and hooks, +and existing role preferences are retained. See the bundled +[orchestrator documentation](skills/orchestrate/README.md) for configuration and +cutover details. + ## Managed Files `ug` backs up files before overwriting them. `ug revert` restores backups. diff --git a/justfile b/justfile index f1fb8558f..a3511b450 100644 --- a/justfile +++ b/justfile @@ -12,11 +12,11 @@ lint: ruff-check ruff-format-check ty # Lint without fixing (CI-equivalent). ruff-check: - uv run ruff check src/ tests/ + uv run ruff check src/ tests/ skills/ # Verify formatting without writing files (CI-equivalent). ruff-format-check: - uv run ruff format --check src/ tests/ + uv run ruff format --check src/ tests/ skills/ # Type-check the package. ty: @@ -24,5 +24,5 @@ ty: # Autofix lint + format in place. Local convenience; not part of the gate. fix: - uv run ruff check --fix src/ tests/ - uv run ruff format src/ tests/ + uv run ruff check --fix src/ tests/ skills/ + uv run ruff format src/ tests/ skills/ diff --git a/skills/orchestrate/scripts/configure.py b/skills/orchestrate/scripts/configure.py new file mode 100644 index 000000000..067a84d8d --- /dev/null +++ b/skills/orchestrate/scripts/configure.py @@ -0,0 +1,438 @@ +#!/usr/bin/env python3 +"""Resolve shared model preferences and manage owned Claude agent definitions.""" + +import argparse +import hashlib +import json +import os +import re +import tempfile +import tomllib +from contextlib import ExitStack, contextmanager +from pathlib import Path + +from ucode.codex_config import ( + DEFAULT_CODEX_CONFIG_PATH, + codex_config_precedence_paths, + codex_managed_config_path, +) +from ucode.os_compatibility.file_lock_cross_os import acquire_exclusive_file_lock, release_file_lock +from ucode.smart_routing.orchestrator import require_enabled + +PACKAGE = Path(__file__).resolve().parents[1] +ROLES = ("explorer", "researcher", "worker", "tester", "reviewer") +EFFORTS = { + "claude": ("low", "medium", "high", "xhigh", "max"), + "codex": ("none", "minimal", "low", "medium", "high", "xhigh", "max"), +} +_CODEX_GATEWAY_PREFIX = "system.ai." +_CODEX_NATIVE_GPT_VERSION = re.compile(r"^(gpt-\d+)\.(\d+)(?=-|$)") +_CODEX_GATEWAY_GPT_VERSION = re.compile(r"^(gpt-\d+)-(\d+)(?=-|$)") + + +def check_path(path): + for part in (path, *path.parents): + if part.is_symlink(): + raise ValueError(f"Refusing symlink: {part}") + if path.exists() and not path.is_file(): + raise ValueError(f"Not a regular file: {path}") + + +def validate_choice(harness, role, choice): + if not isinstance(choice, dict) or set(choice) - {"model", "effort"}: + raise ValueError(f"Invalid choice for {harness}/{role}") + model = choice.get("model") + if ( + not isinstance(model, str) + or not model + or any(character.isspace() or not character.isprintable() for character in model) + ): + raise ValueError(f"Invalid model for {harness}/{role}") + if choice.get("effort") is not None and choice["effort"] not in EFFORTS[harness]: + raise ValueError(f"Invalid effort for {harness}/{role}: expected {EFFORTS[harness]}") + + +def codex_model_candidates(model): + if "/" in model or ":" in model: + return [model] + if model.startswith(_CODEX_GATEWAY_PREFIX): + gateway_slug = model.removeprefix(_CODEX_GATEWAY_PREFIX) + native = _CODEX_GATEWAY_GPT_VERSION.sub(r"\1.\2", gateway_slug, count=1) + return [model, native] if native else [model] + gateway_slug = _CODEX_NATIVE_GPT_VERSION.sub(r"\1-\2", model, count=1) + return [model, _CODEX_GATEWAY_PREFIX + gateway_slug] + + +def active_codex_catalog_path(): + for config_path in codex_config_precedence_paths( + codex_managed_config_path(), DEFAULT_CODEX_CONFIG_PATH + ): + if not config_path.exists(): + continue + try: + with config_path.open("rb") as stream: + configured = tomllib.load(stream).get("model_catalog_json") + except (OSError, TypeError, ValueError) as error: + raise ValueError(f"Invalid Codex configuration: {config_path}") from error + if configured is None: + continue + if not isinstance(configured, str) or not configured: + raise ValueError(f"Invalid model_catalog_json in {config_path}") + path = Path(configured).expanduser() + return (path if path.is_absolute() else config_path.parent / path).resolve() + return None + + +def read_codex_catalog_slugs(path): + try: + catalog = json.loads(path.read_text()) + except (OSError, json.JSONDecodeError, UnicodeError) as error: + raise ValueError(f"Invalid Codex model catalog: {path}") from error + models = catalog.get("models") if isinstance(catalog, dict) else None + if ( + not isinstance(models, list) + or not models + or any( + not isinstance(model, dict) or not isinstance(model.get("slug"), str) + for model in models + ) + ): + raise ValueError(f"Invalid Codex model catalog: {path}") + return {model["slug"] for model in models} + + +def read_config(path): + check_path(path) + try: + data = json.loads(path.read_text()) if path.exists() else {} + except (json.JSONDecodeError, UnicodeError) as error: + raise ValueError( + f"Invalid JSON configuration: {path}; repair or restore it before retrying" + ) from error + if not isinstance(data, dict) or set(data) - {"claude", "codex", "_generated"}: + raise ValueError(f"Invalid configuration keys: {path}") + return data + + +def validate_ownership(data, path): + receipts = data.get("_generated", {}) + if not isinstance(receipts, dict) or set(receipts) - set(ROLES): + raise ValueError(f"Invalid ownership record: {path}") + if any( + not isinstance(receipt, str) or not re.fullmatch(r"[0-9a-f]{64}", receipt) + for receipt in receipts.values() + ): + raise ValueError(f"Invalid ownership hash: {path}") + + +def load_config(path, *, validate_choices=True, harness=None): + data = read_config(path) + for selected_harness in (harness,) if harness is not None else ("claude", "codex"): + roles = data.get(selected_harness, {}) + if not isinstance(roles, dict) or set(roles) - set(ROLES): + raise ValueError(f"Invalid {selected_harness} roles: {path}") + if validate_choices: + for role, choice in roles.items(): + validate_choice(selected_harness, role, choice) + if harness != "codex": + validate_ownership(data, path) + return data + + +def scope_paths(project): + if project is not None: + project = project.expanduser().resolve() + if not project.is_dir(): + raise ValueError(f"Project directory does not exist: {project}") + return project / ".model-orchestrator.json", project / ".claude" / "agents" + config_root = ( + Path(os.environ.get("XDG_CONFIG_HOME", str(Path.home() / ".config"))).expanduser().resolve() + ) + claude_root = ( + Path(os.environ.get("CLAUDE_CONFIG_DIR", str(Path.home() / ".claude"))) + .expanduser() + .resolve() + ) + return config_root / "model-orchestrator" / "config.json", claude_root / "agents" + + +def agent_name(role, project): + return f"model-orchestrator-custom-{'project' if project is not None else 'user'}-{role}" + + +def agent_content(role, choice, project): + text = (PACKAGE / "agents" / f"{role}.md").read_text() + text = re.sub(r"^name: .+$", f"name: {agent_name(role, project)}", text, count=1, flags=re.M) + model = "model: " + json.dumps(choice["model"]) + if choice.get("effort") is not None: + model += "\neffort: " + choice["effort"] + return re.sub(r"^model: .+$", lambda _: model, text, count=1, flags=re.M).encode() + + +def digest(content): + return hashlib.sha256(content).hexdigest() + + +def replace_file(path, content): + if content is None: + path.unlink(missing_ok=True) + return + path.parent.mkdir(parents=True, exist_ok=True) + fd, temporary = tempfile.mkstemp(prefix=".model-orchestrator-", dir=path.parent) + try: + with os.fdopen(fd, "wb") as stream: + stream.write(content) + stream.flush() + os.fsync(stream.fileno()) + os.replace(temporary, path) + finally: + Path(temporary).unlink(missing_ok=True) + + +@contextmanager +def configuration_lock(path, *, read_only_lock_file=False): + check_path(path) + path.parent.mkdir(parents=True, exist_ok=True) + lock = path.with_name(path.name + ".lock") + if lock.is_dir() and not lock.is_symlink(): + raise ValueError(f"Legacy lock directory: {lock}; remove only after its writer exits") + check_path(lock) + mode = "rb" if read_only_lock_file and lock.exists() else "a+b" + with lock.open(mode) as stream: + acquire_exclusive_file_lock(stream) + try: + yield + finally: + release_file_lock(stream) + + +def transaction_paths(project): + config_path, agent_dir = scope_paths(project) + journal = config_path.with_name(config_path.name + ".transaction.json") + paths = {"config": config_path} + paths.update({role: agent_dir / f"{agent_name(role, project)}.md" for role in ROLES}) + return journal, paths + + +def recovery_entry(before, after): + return { + "before": before.hex() if before is not None else None, + "after": after.hex() if after is not None else None, + } + + +def recover_scope(project): + journal, paths = transaction_paths(project) + check_path(journal) + if not journal.exists(): + return + entries = json.loads(journal.read_text()) + if not isinstance(entries, dict) or "config" not in entries or set(entries) - set(paths): + raise ValueError(f"Invalid recovery journal: {journal}") + restores = [] + for name, entry in entries.items(): + if not isinstance(entry, dict) or set(entry) != {"before", "after"}: + raise ValueError(f"Invalid recovery entry: {journal}") + if any(value is not None and not isinstance(value, str) for value in entry.values()): + raise ValueError(f"Invalid recovery content: {journal}") + before = bytes.fromhex(entry["before"]) if entry["before"] is not None else None + after = bytes.fromhex(entry["after"]) if entry["after"] is not None else None + target = paths[name] + check_path(target) + existing = target.read_bytes() if target.exists() else None + if existing not in (before, after): + raise ValueError(f"Preserving edited file during recovery: {target}") + if existing != before: + restores.append((target, before)) + for target, content in reversed(restores): + replace_file(target, content) + journal.unlink() + + +def update_scope(project, previous, desired, *, manage_claude=True): + config_path, agent_dir = scope_paths(project) + changes = {} + receipts = {} + for role in ROLES if manage_claude else (): + path = agent_dir / f"{agent_name(role, project)}.md" + owned = previous.get("_generated", {}).get(role) + choice = desired.get("claude", {}).get(role) + if not owned and choice is None: + continue + check_path(path) + existing = path.read_bytes() if path.exists() else None + if existing is not None and (not owned or digest(existing) != owned): + raise ValueError(f"Preserving unowned or edited agent: {path}") + content = agent_content(role, choice, project) if choice else None + if content is not None: + receipts[role] = digest(content) + if existing != content: + changes[path] = content + if manage_claude: + desired = {key: value for key, value in desired.items() if key != "_generated"} + if receipts: + desired["_generated"] = receipts + check_path(config_path) + changes[config_path] = (json.dumps(desired, indent=2) + "\n").encode() if desired else None + before = {path: path.read_bytes() if path.exists() else None for path in changes} + changes = {path: content for path, content in changes.items() if before[path] != content} + if not changes: + return [] + journal, paths = transaction_paths(project) + check_path(journal) + if journal.exists(): + raise ValueError(f"Pending recovery journal: {journal}; run show before updating") + entries = { + name: recovery_entry(before[path], changes[path]) + for name, path in paths.items() + if path in changes + } + if "config" not in entries: + entries["config"] = recovery_entry(before[config_path], before[config_path]) + replace_file(journal, (json.dumps(entries) + "\n").encode()) + try: + for path, content in changes.items(): + replace_file(path, content) + journal.unlink() + except OSError: + recover_scope(project) + raise + return [str(path) for path in changes] + + +def resolve(harness, project): + require_enabled() + with ExitStack() as locks: + scopes = [] + for scope in [None, project] if project is not None else [None]: + config_path = scope_paths(scope)[0] + journal = transaction_paths(scope)[0] + check_path(journal) + if config_path.exists() or journal.exists(): + locks.enter_context(configuration_lock(config_path, read_only_lock_file=True)) + recover_scope(scope) + scopes.append( + (scope, load_config(config_path, validate_choices=False, harness=harness)) + ) + catalog_path = active_codex_catalog_path() if harness == "codex" else None + return resolve_models(harness, scopes, catalog_path) + + +def resolve_models(harness, scopes, catalog_path=None): + catalog_slugs = read_codex_catalog_slugs(catalog_path) if catalog_path is not None else None + result = {} + for role in ROLES: + choice = ( + {"model": "gpt-5.6-luna", "effort": "max"} + if harness == "codex" + else {"model": "sonnet", "effort": None} + ) + configured = False + subagent_type = f"ug-smart-router:{role}" + for scope, data in reversed(scopes): + if role not in data.get(harness, {}): + continue + validate_choice(harness, role, data[harness][role]) + choice = {"effort": None, **data[harness][role]} + configured = True + if harness == "claude": + subagent_type = agent_name(role, scope) + path = scope_paths(scope)[1] / f"{subagent_type}.md" + check_path(path) + expected = agent_content(role, choice, scope) + if ( + not path.exists() + or path.read_bytes() != expected + or data.get("_generated", {}).get(role) != digest(expected) + ): + raise ValueError( + f"Missing/stale Claude agent; run set for {role} again: {path}" + ) + break + result[role] = dict(choice) + if harness == "claude": + result[role]["subagent_type"] = subagent_type + else: + result[role]["reasoning_effort"] = result[role].pop("effort") + result[role]["allow_inherited_fallback"] = not configured + candidates = codex_model_candidates(result[role]["model"]) + if catalog_slugs is not None and len(candidates) > 1: + # A missing catalog entry is not an invalid preference. Preserve + # the fallback policy while native spawn establishes availability. + result[role]["model"] = next( + (candidate for candidate in candidates if candidate in catalog_slugs), + result[role]["model"], + ) + if harness == "claude" and os.environ.get("CLAUDE_CODE_SUBAGENT_MODEL_FORCE", "").lower() in ( + "1", + "true", + ): + forced = os.environ.get("CLAUDE_CODE_SUBAGENT_MODEL") + if not forced or forced == "inherit": + raise ValueError( + "CLAUDE_CODE_SUBAGENT_MODEL_FORCE selects the parent model; role models cannot be resolved" + ) + if any(choice["model"] != forced for choice in result.values()): + raise ValueError( + "CLAUDE_CODE_SUBAGENT_MODEL_FORCE conflicts with the configured role models" + ) + return result + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("action", choices=("show", "set", "unconfigure")) + parser.add_argument("--harness", choices=tuple(EFFORTS)) + parser.add_argument("--role", choices=ROLES) + parser.add_argument("--model") + parser.add_argument("--effort") + scope = parser.add_mutually_exclusive_group(required=True) + scope.add_argument("--project", type=Path) + scope.add_argument("--user", action="store_true") + args = parser.parse_args() + if args.action == "unconfigure" and args.harness: + parser.error("unconfigure removes the selected scope; omit --harness") + if args.action != "unconfigure" and not args.harness: + parser.error("show/set requires --harness") + if args.action == "set" and (not args.role or not args.model): + parser.error("set requires --role and --model") + if args.action != "set" and any((args.role, args.model, args.effort)): + parser.error("--role, --model, and --effort require set") + result: dict + try: + if args.action == "show": + result = resolve(args.harness, args.project) + else: + path = scope_paths(args.project)[0] + with configuration_lock(path): + recover_scope(args.project) + if args.action == "unconfigure": + previous = read_config(path) + validate_ownership(previous, path) + else: + previous = load_config( + path, validate_choices=args.harness != "codex", harness=args.harness + ) + desired: dict = json.loads(json.dumps(previous)) if args.action == "set" else {} + if args.action == "set": + desired.setdefault(args.harness, {})[args.role] = { + "model": args.model, + "effort": args.effort, + } + validate_choice(args.harness, args.role, desired[args.harness][args.role]) + result = { + "changed": update_scope( + args.project, previous, desired, manage_claude=args.harness != "codex" + ) + } + if args.harness == "claude" and result["changed"]: + result["next"] = ( + "Restart Claude Code after initial setup; run show to verify the resolved map." + ) + print(json.dumps(result, indent=2)) + except (OSError, ValueError) as error: + parser.exit(1, f"{error}\n") + + +if __name__ == "__main__": + main() diff --git a/skills/smart-router/SKILL.md b/skills/smart-router/SKILL.md index b21e2680a..07118d7b7 100644 --- a/skills/smart-router/SKILL.md +++ b/skills/smart-router/SKILL.md @@ -1,9 +1,9 @@ --- name: smart-router -description: Enable or disable Unity Gateway subagent model routing for the current smart-routed Claude or Codex session. +description: Enable or disable Unity Gateway subagent model routing and automatic orchestration together for the current smart-routed Claude or Codex session. allowed-tools: Bash("$UCODE_SMART_ROUTER_PYTHON" -m ucode.cli claude --enable-smart-routing), Bash("$UCODE_SMART_ROUTER_PYTHON" -m ucode.cli claude --disable-smart-routing), Bash("$UCODE_SMART_ROUTER_PYTHON" -m ucode.cli codex --enable-smart-routing), Bash("$UCODE_SMART_ROUTER_PYTHON" -m ucode.cli codex --disable-smart-routing) metadata: - version: "1.1.0" + version: "1.2.0" --- # Smart Router @@ -22,5 +22,10 @@ If `UCODE_SMART_ROUTER_PYTHON` or `UCODE_SESSION_ENV_FILE` is unset, ask the use to restart through an updated Unity Gateway with smart routing enabled. With no argument, explain that only `on` and `off` are accepted. Do not edit the state file. -This affects subsequent subagent model selection in the current session, not the root model or -first prompt. Return the command's result. +This affects subsequent subagent model selection and automatic orchestration in the current +session, not the root model or first prompt. When turned off, earlier orchestrate instructions +are superseded: do not start new automatic delegation or fall back to orchestrator role models. +Continue in the root unless the user explicitly requests a subagent. Honor that request using +native tools and normal harness model selection, without the orchestrator or its resolution +helper; keep routing off. Existing children can finish. When turned on, apply the orchestrate +skill to further work. Return the command's result. diff --git a/src/ucode/agents/claude.py b/src/ucode/agents/claude.py index 68e3f1d00..5b459f9cc 100644 --- a/src/ucode/agents/claude.py +++ b/src/ucode/agents/claude.py @@ -82,6 +82,7 @@ external_provider_selected, ) from ucode.os_compatibility import subprocess_cross_os +from ucode.smart_routing import orchestrator from ucode.smart_routing import v2 as smart_routing_v2 from ucode.smart_routing.claude_hooks import ( FIRST_PROMPT_SOCKET_ENV, @@ -1857,7 +1858,9 @@ def _compose_v2_settings(tool_args: list[str]) -> tuple[dict, list[str]]: settings: dict = {} for value in caller_values: settings = _merge_claude_settings(settings, _load_caller_settings(value)) - return _merge_claude_settings(settings, read_json_safe(CLAUDE_SETTINGS_PATH)), remaining + settings = _merge_claude_settings(settings, read_json_safe(CLAUDE_SETTINGS_PATH)) + orchestrator.suppress_legacy_claude_plugin(settings, CLAUDE_USER_SETTINGS_PATH) + return settings, remaining def _launch_model_args(tool_args: list[str], launch_model: str | None) -> list[str]: @@ -1972,10 +1975,6 @@ def _build_claude_argv( """ source_args = ["--setting-sources", _RELAYED_SETTING_SOURCES] if relayed else [] caller_values, remaining = _extract_caller_settings(tool_args) - if not caller_values and settings_override is None: - # No caller --settings: hand Claude ucode's settings file directly (the - # common path; behavior unchanged). - return [binary, *source_args, "--settings", str(CLAUDE_SETTINGS_PATH), *tool_args] caller_settings: dict = {} for value in caller_values: caller_settings = _merge_claude_settings(caller_settings, _load_caller_settings(value)) @@ -1984,6 +1983,9 @@ def _build_claude_argv( merged = _merge_claude_settings(caller_settings, read_json_safe(CLAUDE_SETTINGS_PATH)) if settings_override is not None: merged = _merge_claude_settings(merged, settings_override) + suppressed = orchestrator.suppress_legacy_claude_plugin(merged, CLAUDE_USER_SETTINGS_PATH) + if not caller_values and settings_override is None and not suppressed: + return [binary, *source_args, "--settings", str(CLAUDE_SETTINGS_PATH), *tool_args] merged_env = merged.get("env") if isinstance(merged_env, dict): merged_env.pop("CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY", None) diff --git a/src/ucode/agents/codex.py b/src/ucode/agents/codex.py index 3c04a93ff..cadbcf09a 100644 --- a/src/ucode/agents/codex.py +++ b/src/ucode/agents/codex.py @@ -21,6 +21,7 @@ codex_config_args, codex_config_precedence_paths, codex_managed_config_path, + codex_working_directory, custom_catalog_models, ) from ucode.config_io import ( @@ -70,6 +71,7 @@ revert_managed_file, ) from ucode.os_compatibility import subprocess_cross_os +from ucode.smart_routing import orchestrator from ucode.smart_routing import v2 as smart_routing_v2 from ucode.smart_routing.codex_hooks import ( remove_smart_routing_hooks, @@ -1153,6 +1155,9 @@ def launch( if workspace: token = _launch_token(state, workspace) os.environ["OAUTH_TOKEN"] = token + legacy_plugin_config = orchestrator.legacy_codex_plugin_config( + CODEX_CONFIG_PATH, cwd=codex_working_directory(tool_args) + ) if _use_legacy_layout(): print_warning_err( f"Codex {agent_version(binary)} is outdated. Upgrade Codex to " @@ -1161,7 +1166,7 @@ def launch( ) _run_codex( state, - [binary, "--profile", CODEX_PROFILE_NAME], + [binary, "--profile", CODEX_PROFILE_NAME, *codex_config_args(legacy_plugin_config)], tool_args, otel_tracing=otel_tracing, workspace=workspace, @@ -1177,6 +1182,8 @@ def launch( f"Cannot launch Codex with the ucode profile because {CODEX_CONFIG_PATH} " "is missing or empty. Run `ucode configure --agents codex` first." ) + # Repeated CLI keys replace earlier overrides, so preserve the profile's other plugins here. + deep_merge_dict(profile_doc, legacy_plugin_config) _set_provider_header(profile_doc, provider) _set_parent_schema_header(profile_doc, parent_schema if not provider else None) updating = tool_args[:1] == ["update"] diff --git a/src/ucode/cli.py b/src/ucode/cli.py index 656abd746..20a85c8f5 100644 --- a/src/ucode/cli.py +++ b/src/ucode/cli.py @@ -139,6 +139,7 @@ from ucode.smart_routing.claude_hooks import FIRST_PROMPT_SOCKET_ENV, ROUTE_FIRST_PROMPT_EVENT from ucode.smart_routing.session_env import ( effective_environment, + fresh_launch, session_env_path, set_session_environment, ) @@ -2310,6 +2311,12 @@ def _toggle_current_smart_routing_session(enabled: bool | None) -> bool: print_err(str(exc)) raise typer.Exit(1) from None print_success(f"Smart Router is {'on' if enabled else 'off'} for this session") + if enabled: + print_note("Automatic orchestration is on; apply the orchestrate skill to further work.") + else: + from ucode.smart_routing.orchestrator import DISABLED_CONTEXT + + print_note(DISABLED_CONTEXT) return True @@ -2937,8 +2944,11 @@ def _launch_tool( provider=provider, ) print_success(f"Starting {TOOL_SPECS[tool]['display']}") - with _smart_routing_v2_flag( - True if managed_smart_routing_enabled and smart_routing_enabled else None + with ( + _smart_routing_v2_flag( + True if managed_smart_routing_enabled and smart_routing_enabled else None + ), + fresh_launch(), ): launch_agent(tool, state, ctx.args, options=launch_options) except RuntimeError as exc: diff --git a/src/ucode/codex_config.py b/src/ucode/codex_config.py index b93148f5d..f9af79907 100644 --- a/src/ucode/codex_config.py +++ b/src/ucode/codex_config.py @@ -2,8 +2,14 @@ from __future__ import annotations +import json import os +import queue +import subprocess +import threading +import time from collections.abc import Mapping +from contextlib import suppress from enum import StrEnum from pathlib import Path @@ -12,10 +18,122 @@ from ucode.config_io import read_json_safe, read_toml_safe from ucode.managed_files import OS, current_os +from ucode.os_compatibility import subprocess_cross_os from ucode.ui import print_warning CODEX_PROFILE_NAME = "ucode" DEFAULT_CODEX_CONFIG_PATH = Path.home() / ".codex" / f"{CODEX_PROFILE_NAME}.config.toml" +CONFIG_READ_TIMEOUT_SECONDS = 15 + + +def codex_working_directory(tool_args: list[str]) -> Path: + """Resolve the launch directory before looking up project configuration.""" + directory = Path.cwd() + args = iter(tool_args) + for arg in args: + if arg == "--": + break + if arg in {"--cd", "-C"}: + value = next(args, None) + if value is not None: + directory = Path(value).expanduser() + elif arg.startswith("--cd="): + directory = Path(arg.partition("=")[2]).expanduser() + elif arg.startswith("-C"): + directory = Path(arg[2:].removeprefix("=")).expanduser() + return directory.resolve() + + +def codex_cli_config_args(tool_args: list[str]) -> list[str]: + """Keep caller configuration overrides, including project trust, in the native lookup.""" + config_args = [] + args = iter(tool_args) + options = {"-c", "--config", "--enable", "--disable"} + for arg in args: + if arg == "--": + break + if arg in options: + value = next(args, None) + if value is not None: + config_args.extend([arg, value]) + elif arg.partition("=")[0] in options or arg.startswith("-c"): + config_args.append(arg) + return config_args + + +def read_effective_codex_config(binary: str, *, cwd: Path, config_args: list[str]) -> dict: + """Let Codex apply its configuration precedence and project trust rules.""" + error = ( + "Could not read Codex configuration for smart routing. Check your Codex configuration " + "or launch with --disable-smart-routing." + ) + try: + process = subprocess_cross_os.popen( + [binary, "app-server", *config_args, "--listen", "stdio://"], + cwd=cwd, + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, + text=True, + ) + except OSError as exc: + raise RuntimeError(error) from exc + stdin, stdout = process.stdin, process.stdout + assert stdin is not None and stdout is not None + messages: queue.Queue = queue.Queue() + + def read_messages() -> None: + try: + for line in stdout: + messages.put(json.loads(line)) + except (OSError, ValueError): + pass + finally: + messages.put(None) + + reader = threading.Thread(target=read_messages, daemon=True) + reader.start() + deadline = time.monotonic() + CONFIG_READ_TIMEOUT_SECONDS + + def request(request_id: int, method: str, params: dict) -> dict: + stdin.write(json.dumps({"id": request_id, "method": method, "params": params}) + "\n") + stdin.flush() + while time.monotonic() < deadline: + message = messages.get(timeout=max(0, deadline - time.monotonic())) + if not isinstance(message, dict): + raise RuntimeError(error) + if message.get("id") == request_id: + result = message.get("result") + if not isinstance(result, dict) or "error" in message: + raise RuntimeError(error) + return result + raise RuntimeError(error) + + try: + request(1, "initialize", {"clientInfo": {"name": "unity-gateway", "version": "1"}}) + stdin.write('{"method":"initialized","params":{}}\n') + stdin.flush() + result = request(2, "config/read", {"cwd": str(cwd), "includeLayers": False}) + config = result.get("config") + if not isinstance(config, dict): + raise RuntimeError(error) + return config + except (OSError, queue.Empty) as exc: + raise RuntimeError(error) from exc + finally: + with suppress(OSError): + stdin.close() + try: + process.wait(timeout=5) + except subprocess.TimeoutExpired: + process.terminate() + try: + process.wait(timeout=5) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=5) + reader.join(timeout=5) + stdout.close() class ModelVisibility(StrEnum): @@ -111,6 +229,9 @@ def _toml_item(value: object) -> Item: if isinstance(value, Mapping): inline = tomlkit.inline_table() for key, child in value.items(): + # Native config/read includes null optional fields; TOML represents them by omission. + if child is None: + continue inline[str(key)] = _toml_item(child) return inline if isinstance(value, list): diff --git a/src/ucode/skills.py b/src/ucode/skills.py index 5f80137ad..d8c24cc65 100644 --- a/src/ucode/skills.py +++ b/src/ucode/skills.py @@ -12,6 +12,7 @@ _LEGACY_SKILL_ROOTS = (".agents/skills",) _SKILL_NAME_PATTERN = re.compile(r"[a-z0-9]+(?:-[a-z0-9]+)*") SMART_ROUTER_SKILL = "smart-router" +ORCHESTRATOR_SKILL = "orchestrate" def _skills_source() -> Path: diff --git a/src/ucode/smart_routing/orchestrator.py b/src/ucode/smart_routing/orchestrator.py new file mode 100644 index 000000000..64cd1c711 --- /dev/null +++ b/src/ucode/smart_routing/orchestrator.py @@ -0,0 +1,164 @@ +"""Activate the bundled orchestrator only in an enabled smart-routing session.""" + +from __future__ import annotations + +import json +import os +import shlex +import shutil +import subprocess +import sys +from collections.abc import Mapping +from pathlib import Path + +from ucode import codex_config, skills +from ucode.config_io import read_json_safe, read_toml_safe +from ucode.smart_routing.hooks import sync_managed_hooks +from ucode.smart_routing.session_env import effective_environment, session_env_path + +HOOK_MODULE = "ucode.smart_routing.orchestrator" +DISABLED_CONTEXT = ( + "UG automatic orchestration is off because smart routing is off for this session. " + "This supersedes any earlier model-orchestrator workflow: do not start new automatic " + "delegation or fall back to orchestrator role models. Explicit user requests for subagents " + "still use native tools and normal harness model selection, without the orchestrator " + "or its model-resolution helper; keep routing off. Otherwise continue the task in the root. " + "Collect results from children already running." +) + + +def _legacy_plugin_ids(plugins: object) -> set[str]: + if not isinstance(plugins, dict): + return set() + return { + name + for name in plugins + if isinstance(name, str) and name.partition("@")[0] == "model-orchestrator" + } + + +def suppress_legacy_claude_plugin(settings: dict, user_settings_path: Path) -> bool: + """Suppress the installed predecessor for this launch, including when routing is off.""" + config_dir = Path(os.environ.get("CLAUDE_CONFIG_DIR", user_settings_path.parent)).expanduser() + installed = read_json_safe(config_dir / "plugins" / "installed_plugins.json") + user_settings = read_json_safe(config_dir / user_settings_path.name) + names = ( + _legacy_plugin_ids(installed.get("plugins")) + | _legacy_plugin_ids(user_settings.get("enabledPlugins")) + | _legacy_plugin_ids(settings.get("enabledPlugins")) + ) + if not names: + return False + plugins = settings.setdefault("enabledPlugins", {}) + if not isinstance(plugins, dict): + raise RuntimeError("Claude settings 'enabledPlugins' must be an object.") + plugins.update(dict.fromkeys(sorted(names), False)) + return True + + +def legacy_codex_plugin_config( + profile_path: Path | None = None, *, cwd: Path | None = None +) -> dict: + """Return a launch override; Isaac owns and may regenerate the saved plugin entries.""" + names: set[str] = set() + directory = (cwd or Path.cwd()).resolve() + paths = codex_config.codex_config_precedence_paths( + codex_config.codex_managed_config_path(), + profile_path or codex_config.DEFAULT_CODEX_CONFIG_PATH, + ) + # Only collect IDs to disable; never promote commands from untrusted project files. + project_paths = [] + for parent in (directory, *directory.parents): + project_paths.append(parent / ".codex" / "config.toml") + if (parent / ".git").exists(): + break + for path in (*paths, *project_paths): + names.update(_legacy_plugin_ids(read_toml_safe(path).get("plugins"))) + # Codex merges this table with the saved map. A dotted key cannot safely encode + # arbitrary marketplace names, and quoted dotted segments are treated literally. + return {"plugins": {name: {"enabled": False} for name in sorted(names)}} if names else {} + + +def enabled(env: Mapping[str, str] | None = None) -> bool: + from ucode.smart_routing.v2 import smart_routing_enabled + + source = os.environ if env is None else env + if source.get("ISAAC_LAUNCH_MODE", "").strip().lower() == "omni": + return False + try: + # The marker is created only after UG selects a supported routing launch. + if not session_env_path(source).is_file(): + return False + except (RuntimeError, OSError): + return False + return smart_routing_enabled(effective_environment(source)) + + +def require_enabled() -> None: + if not enabled(): + raise ValueError(DISABLED_CONTEXT) + + +def skill_directory() -> Path: + return skills._skills_source() / skills.ORCHESTRATOR_SKILL + + +def add_claude_agents(plugin_dir: Path) -> None: + """Load roles alongside the router's exact-model agents, only for this launch.""" + shutil.copytree(skill_directory() / "agents", plugin_dir / "agents", dirs_exist_ok=True) + + +def sync_hooks(doc: dict, *, agent: str) -> None: + argv = [sys.executable, "-m", HOOK_MODULE] + hook = { + "type": "command", + "command": shlex.join(argv), + "timeout": 5, + } + if agent == "codex": + hook["command_windows"] = subprocess.list2cmdline(argv) + sync_managed_hooks( + doc, + HOOK_MODULE, + { + "UserPromptSubmit": [{"hooks": [hook]}], + "SessionStart": [{"matcher": "compact", "hooks": [hook]}], + }, + ) + + +def hook_output(payload: object) -> dict | None: + if not isinstance(payload, dict) or payload.get("agent_id"): + return None + event = payload.get("hook_event_name") + if event != "UserPromptSubmit" and not ( + event == "SessionStart" and payload.get("source") == "compact" + ): + return None + context = DISABLED_CONTEXT + if enabled(): + directory = skill_directory() + try: + workflow = (directory / "SKILL.md").read_text(encoding="utf-8") + except (OSError, UnicodeError): + return None + context = ( + "Apply the UG model-orchestrator workflow to this task. " + "Smart routing and automatic orchestration share the same session controls.\n" + f"Skill directory: {directory}\n\n{workflow}" + ) + return {"hookSpecificOutput": {"hookEventName": event, "additionalContext": context}} + + +def main() -> None: + try: + payload = json.load(sys.stdin) + except (OSError, UnicodeError, ValueError): + return + output = hook_output(payload) + if output is not None: + print(json.dumps(output)) + + +if __name__ == "__main__": + main() diff --git a/src/ucode/smart_routing/session_env.py b/src/ucode/smart_routing/session_env.py index 21887c945..a134e460a 100644 --- a/src/ucode/smart_routing/session_env.py +++ b/src/ucode/smart_routing/session_env.py @@ -6,7 +6,8 @@ import os import sys import tempfile -from collections.abc import Mapping, MutableMapping +from collections.abc import Iterator, Mapping, MutableMapping +from contextlib import contextmanager from pathlib import Path from ucode.config_io import atomic_write_json @@ -17,6 +18,21 @@ _ALLOWED_KEYS = frozenset(SMART_ROUTING_ENV_KEYS) +@contextmanager +def fresh_launch() -> Iterator[None]: + """Prevent a nested, non-routed launch from inheriting its parent's eligibility.""" + keys = (SESSION_ENV_VAR, SESSION_PYTHON_ENV_VAR) + previous = {key: os.environ.pop(key, None) for key in keys} + try: + yield + finally: + for key, value in previous.items(): + if value is None: + os.environ.pop(key, None) + else: + os.environ[key] = value + + def start_session(env: MutableMapping[str, str] | None = None) -> Path: """Create an empty override file and expose it to the launched harness.""" target = os.environ if env is None else env diff --git a/src/ucode/smart_routing/v2.py b/src/ucode/smart_routing/v2.py index 50790dfac..e8eaf299c 100644 --- a/src/ucode/smart_routing/v2.py +++ b/src/ucode/smart_routing/v2.py @@ -10,20 +10,23 @@ import time import urllib.request from collections.abc import Callable, MutableMapping +from copy import deepcopy from pathlib import Path from tempfile import TemporaryDirectory from typing import NoReturn, TextIO from ucode import config_io from ucode.codex_config import ( + codex_cli_config_args, codex_config_args, + codex_working_directory, custom_catalog_models, custom_catalog_path, + read_effective_codex_config, ) from ucode.config_io import ( APP_DIR, read_json_safe, - read_toml_safe, write_json_file, write_text_file, ) @@ -47,8 +50,8 @@ acquire_exclusive_file_lock, release_file_lock, ) -from ucode.skills import SMART_ROUTER_SKILL, install_skill -from ucode.smart_routing import claude_routing, codex_interposer, routing +from ucode.skills import ORCHESTRATOR_SKILL, SMART_ROUTER_SKILL, install_skill +from ucode.smart_routing import claude_routing, codex_interposer, orchestrator, routing from ucode.smart_routing.claude_hooks import ( FIRST_PROMPT_SOCKET_ENV, sync_first_prompt_hook, @@ -84,10 +87,11 @@ class ClaudeRoutingSetupError(RuntimeError): def _prepare_smart_router_session(agent: str) -> Path: - try: - install_skill(SMART_ROUTER_SKILL, agent, config_io.APP_DIR.parent) - except (OSError, RuntimeError) as exc: - print_warning(f"Could not install the Smart Router skill: {exc}") + for skill in (SMART_ROUTER_SKILL, ORCHESTRATOR_SKILL): + try: + install_skill(skill, agent, config_io.APP_DIR.parent) + except (OSError, RuntimeError) as exc: + print_warning(f"Could not install the {skill} skill: {exc}") return start_session() @@ -332,6 +336,7 @@ def _write_routed_claude_plugin(plugin_dir: Path, model_ids: list[str]) -> None: ] ), ) + orchestrator.add_claude_agents(plugin_dir) def _request_claude_routing_decision( @@ -538,6 +543,7 @@ def launch_claude( sync_smart_routing_hooks(settings, routing_state, enabled=True) if route_first_prompt: sync_first_prompt_hook(settings, hook_executable) + orchestrator.sync_hooks(settings, agent="claude") model_setting = _ClaudeModelSettingGuard(user_settings_path) def route_prompt(prompt: str) -> claude_pty.FirstPromptRoute: @@ -603,22 +609,25 @@ def _cached_routing_models(state: dict) -> list[str]: return routing_models(state) -def _codex_home_config_path() -> Path: - codex_home = os.environ.get("CODEX_HOME") - if codex_home: - return Path(codex_home).expanduser() / "config.toml" - return Path.home() / ".codex" / "config.toml" - - -def _v2_pre_tool_use_hooks(state: dict, available_models: list[str]) -> list[dict]: - doc = read_toml_safe(_codex_home_config_path()) - configured_hooks = doc.get("hooks") - existing = configured_hooks.get("PreToolUse") if isinstance(configured_hooks, dict) else None - return merge_pre_tool_use_hooks( +def _v2_hooks(state: dict, available_models: list[str], config: dict) -> dict: + configured_hooks = config.get("hooks") + events = ("PreToolUse", "UserPromptSubmit", "SessionStart") + # Only override events UG changes. Codex retains ownership of every other event. + doc = { + "hooks": { + event: deepcopy(configured_hooks[event]) + for event in events + if isinstance(configured_hooks, dict) and event in configured_hooks + } + } + existing = doc["hooks"].get("PreToolUse") + doc["hooks"]["PreToolUse"] = merge_pre_tool_use_hooks( existing if isinstance(existing, list) else [], state, available_models=available_models, ) + orchestrator.sync_hooks(doc, agent="codex") + return doc["hooks"] def launch_codex( @@ -659,19 +668,26 @@ def launch_codex( catalog_path = custom_catalog_path() if catalog_path is not None: overlay["model_catalog_json"] = str(catalog_path) - overlay["hooks"] = { - "PreToolUse": _v2_pre_tool_use_hooks(state, available_models), - } - session_env_path = _prepare_smart_router_session("codex") + cwd = codex_working_directory(tool_args) + config = read_effective_codex_config( + binary, + cwd=cwd, + config_args=[*codex_config_args(overlay), *codex_cli_config_args(tool_args)], + ) + overlay["hooks"] = _v2_hooks(state, available_models, config) + overlay["features.hooks"] = True + legacy_plugin_config = orchestrator.legacy_codex_plugin_config(cwd=cwd) + overlay.update(legacy_plugin_config) + _prepare_smart_router_session("codex") # Codex constructs tool subprocess environments through its shell policy. - # Pass both the session marker and its launching interpreter through that policy. - overlay[f"shell_environment_policy.set.{SESSION_ENV_VAR}"] = str(session_env_path) - overlay[f"shell_environment_policy.set.{SESSION_PYTHON_ENV_VAR}"] = os.environ[ - SESSION_PYTHON_ENV_VAR - ] + # The skill's gate needs the same launch baseline as the routing hook, even + # when the user's policy filters inherited environment variables. + for key in (SESSION_ENV_VAR, SESSION_PYTHON_ENV_VAR, *SMART_ROUTING_ENV_KEYS): + if key in os.environ: + overlay[f"shell_environment_policy.set.{key}"] = os.environ[key] config_args = codex_config_args(overlay) if not first_prompt_routing_enabled(): - # Subagent-only routing needs neither the app-server nor the interposer: + # Subagent-only routing needs no persistent app-server or interposer: # the hooks ride in the CLI config, so launch the TUI directly. exec_or_spawn([binary, *config_args, *tool_args]) app_port = _free_port() @@ -713,7 +729,16 @@ def launch_codex( } ) tui = subprocess_cross_os.popen( - [binary, *provider_args, "--remote", tui_url, "--model", start_model, *tool_args] + [ + binary, + *provider_args, + *codex_config_args(legacy_plugin_config), + "--remote", + tui_url, + "--model", + start_model, + *tool_args, + ] ) try: returncode = tui.wait() diff --git a/tests/README.md b/tests/README.md index 66d24fa4f..50c97de01 100644 --- a/tests/README.md +++ b/tests/README.md @@ -131,6 +131,27 @@ that Claude settings and Codex's shell policy carry the interpreter and session These are component checks; they do not establish native skill permission matching or PowerShell execution. +`test_orchestrator.py` covers shared routing state, off/on transitions, suppression +of retained skills outside eligible sessions, root prompts and compaction, +user-hook preservation, and bundled Claude roles. `test_orchestrator_config.py` +ports the plugin's preference, ownership, locking, and interrupted-write recovery +coverage and checks UG catalog precedence. Real subprocess tests verify that preference +reads, writes, and removal wait for an existing writer, preserve files while waiting, +and complete after the lock is released. `test_orchestrator_legacy_plugins.py` +and the Claude/Codex launcher tests check native per-launch overrides that disable +legacy marketplace registrations with routing on or off, including Codex's +app-server and remote TUI. They check config discovery, unrelated-plugin and hook +preservation, and unchanged saved settings, including nested project registrations +and `--cd`. `test_codex_config.py` exercises the native configuration protocol, +timeout cleanup, and optional hook fields; launch tests preserve its resolved +handlers and leave unrelated hook events to Codex. These are component assertions. +The toggle integration journeys require both bundled skills, verify the saved +session controls and native tool-result confirmation after each toggle, and +explicitly request their children, including while routing is off. Live automatic +delegation and legacy-hook execution are not covered by these tests. +`test_integration_evidence.py` checks native tool-result extraction for both agents, +including collapsed-output records, and excludes user echoes and assistant claims. + The portable Windows routing test checks native executable forwarding, generated hooks/plugins, caller arguments, and cleanup without Unix imports. It does not establish live Windows hook execution or interactive routing. diff --git a/tests/conftest.py b/tests/conftest.py index 2b0c9f090..1ac6d2077 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -29,12 +29,14 @@ def _isolate_ucode_state(tmp_path, monkeypatch): it can never touch the developer's real ~/.ucode/state.json or invoke the privileged writer for an OS-managed agent config. """ + import ucode.codex_config as codex_config_mod import ucode.config_io as config_io_mod import ucode.databricks as databricks_mod import ucode.managed_config as managed_config_mod import ucode.managed_files as managed_files_mod import ucode.os_compatibility.subprocess_cross_os as subprocess_cross_os_mod import ucode.state as state_mod + from ucode.agents import claude as claude_mod from ucode.agents import codex as codex_mod state_dir = tmp_path / ".ucode" @@ -56,6 +58,12 @@ def _isolate_ucode_state(tmp_path, monkeypatch): codex_mod, "CODEX_MODEL_CATALOG_PATH", state_dir / "codex-model-catalog.json" ) monkeypatch.setattr(codex_mod, "CODEX_CONFIG_PATH", tmp_path / ".codex" / "ucode.config.toml") + # Launch-time plugin discovery must not read the developer's installed plugins. + monkeypatch.setattr(codex_config_mod, "DEFAULT_CODEX_CONFIG_PATH", codex_mod.CODEX_CONFIG_PATH) + monkeypatch.setattr(codex_config_mod, "codex_managed_config_path", lambda: None) + monkeypatch.setattr(claude_mod, "CLAUDE_USER_SETTINGS_PATH", tmp_path / ".claude/settings.json") + monkeypatch.delenv("CLAUDE_CONFIG_DIR", raising=False) + monkeypatch.delenv("CODEX_HOME", raising=False) def reject_privileged_write(path, _desired_text): pytest.fail( diff --git a/tests/integration/README.md b/tests/integration/README.md index 2cb2bcbbb..5d81bfefb 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -281,6 +281,24 @@ PATH conflicts for the Smart Router skill have subprocess/component coverage in `ug` first in PATH. The live journeys above do not inject a second installation or establish PowerShell command execution. +The toggle journeys require both bundled skills (`orchestrate` and `smart-router`) +to be installed. They verify the saved session controls, a new CLI confirmation in +the native tool-result records, and a new assistant answer after each skill invocation. +Collapsed terminal output is allowed; the answer need not repeat the CLI's exact wording. +Each following child still verifies whether a routing decision occurred. +Their off-phase child is an explicit user-requested delegation; +these journeys do not establish automatic orchestration behavior. Shared on/off +state, root-only activation, compaction, retained skills, and role preference +compatibility are covered in `../test_orchestrator.py` and +`../test_orchestrator_config.py`. `../test_orchestrator_legacy_plugins.py` and +launcher component tests check per-launch suppression of installed legacy plugins, +including non-routed launches, nested project config and `--cd`, and preservation +of saved settings and unrelated plugins/hooks. Native config protocol and timeout +handling have component coverage in `../test_codex_config.py`; native project trust +and hook execution are not exercised by this integration suite. Live automatic +delegation and legacy-hook execution remain +unverified by this suite. + The portable `../test_claude_windows_smart_routing.py` checks the Windows subagent-only fallback without Unix imports. Native Windows TUI and hook execution remain outside this integration suite. diff --git a/tests/integration/test_ug_smart_routing_hooks.py b/tests/integration/test_ug_smart_routing_hooks.py index 58ddcffe3..7fc8ecb8e 100644 --- a/tests/integration/test_ug_smart_routing_hooks.py +++ b/tests/integration/test_ug_smart_routing_hooks.py @@ -9,14 +9,18 @@ """ import json +from pathlib import Path import pytest from utils.constants import CLAUDE_SMART_ROUTING_MODELS, CODEX_SMART_ROUTING_MODELS from utils.evidence import ( SubagentCalculation, + agent_sessions, assert_subagent_routed, - assistant_answer_contains, + assistant_answers, + is_child_session, read_jsonl, + tool_outputs, ) from utils.managed import ( build_claude_agent_config, @@ -121,21 +125,46 @@ def _toggle_with_skill(tui, session, agent: str, enabled: bool) -> None: for path in skill_root.iterdir() if path.is_dir() and path.name not in ignored_skills ) - assert installed_skills == ["smart-router"], installed_skills + assert installed_skills == ["orchestrate", "smart-router"], installed_skills state = "on" if enabled else "off" invocation = f"/smart-router {state}" if agent == "claude" else f"$smart-router {state}" - confirmation = f"{state} for this session" + controls = list(Path(session.env["TMPDIR"]).glob("ug-session-env-*/env.json")) + assert len(controls) == 1, controls + expected = ( + {} + if enabled + else { + "ENABLE_SMART_ROUTING_V2": "0", + "ENABLE_SMART_ROUTING_SUBAGENT_ONLY": "0", + } + ) + assert json.loads(controls[0].read_text()) != expected + + confirmation = f"Smart Router is {state} for this session" + + def completion_counts(): + answers = confirmations = 0 + for path, records in agent_sessions(session, agent).items(): + if is_child_session(agent, path, records): + continue + answers += len(assistant_answers(agent, records)) + confirmations += sum(confirmation in output for output in tool_outputs(agent, records)) + return answers, confirmations + + before_answers, before_confirmations = completion_counts() tui.submit(invocation) + + def toggled(_screen): + answers, confirmations = completion_counts() + return ( + json.loads(controls[0].read_text()) == expected + and confirmations > before_confirmations + and answers > before_answers + ) + tui.wait_for( - lambda screen: ( - "Smart Router" in screen - and confirmation in screen - and ( - assistant_answer_contains(session, agent, confirmation) - or assistant_answer_contains(session, agent, f"**{state}** for this session") - ) - ), + toggled, f"the installed Smart Router skill to turn routing {state}", timeout=120, ) @@ -255,12 +284,15 @@ def test_smart_router_skill_toggles_claude_subagent_routing(live_session, worksp installed Smart Router skill to turn routing off, spawn another child, turn routing back on through the skill, and spawn a third child in the same real TUI session. - Expected: Smart Router is the only user-installed Claude skill; all three uniquely tagged + Expected: Smart Router and orchestrate are the only user-installed Claude skills; + each invocation records the CLI confirmation in the native transcript and changes the saved + routing controls, even with collapsed terminal output; all three uniquely tagged calculations complete in native child sessions; only the first and third show the subagent-routing banner and produce live gateway decisions correlated with those children. No first-prompt routing wrapper starts. """ session = live_session + session.env["TMPDIR"] = str(tmp_path) session.env["ENABLE_SMART_ROUTING_V2"] = "1" session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" config = build_coding_agent_config( @@ -301,12 +333,15 @@ def test_smart_router_skill_toggles_codex_subagent_routing(live_session, workspa installed Smart Router skill to turn routing off, spawn another child, turn routing back on through the skill, and spawn a third child in the same real TUI session. - Expected: Smart Router is the only user-installed Codex skill; all three uniquely tagged + Expected: Smart Router and orchestrate are the only user-installed Codex skills; + each invocation records the CLI confirmation in the native transcript and changes the saved + routing controls, even with collapsed terminal output; all three uniquely tagged calculations complete in native child sessions; only the first and third show the subagent-routing banner and produce live gateway decisions correlated with those children. No first-prompt interposer starts. """ session = live_session + session.env["TMPDIR"] = str(tmp_path) session.env["ENABLE_SMART_ROUTING_V2"] = "1" session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" config = build_coding_agent_config( diff --git a/tests/integration/utils/evidence.py b/tests/integration/utils/evidence.py index 938fb7eba..d3f77a893 100644 --- a/tests/integration/utils/evidence.py +++ b/tests/integration/utils/evidence.py @@ -52,6 +52,38 @@ def assistant_answers(agent: str, records: list[dict]) -> list[str]: return helper.assistant_answers(records) if helper is not None else [] +def tool_outputs(agent: str, records: list[dict]) -> list[str]: + """Read native tool results even when their terminal output is collapsed.""" + contents = [] + for record in records: + if agent == "claude" and record.get("type") == "user": + contents.extend( + part.get("content") + for part in record.get("message", {}).get("content", []) + if isinstance(part, dict) + and part.get("type") == "tool_result" + and not part.get("is_error") + ) + if agent == "codex" and record.get("type") == "response_item": + payload = record.get("payload", {}) + if payload.get("type") in {"function_call_output", "custom_tool_call_output"}: + contents.append(payload.get("output")) + + outputs = [] + for content in contents: + if isinstance(content, str): + outputs.append(content) + elif isinstance(content, list): + outputs.extend( + part["text"] + for part in content + if isinstance(part, dict) + and part.get("type") in {"text", "input_text"} + and isinstance(part.get("text"), str) + ) + return outputs + + def is_child_session(agent: str, path: str, records: list[dict]) -> bool: return _AGENT_HELPERS.get(agent, codex).is_child_session(path, records) diff --git a/tests/test_agent_claude.py b/tests/test_agent_claude.py index e9bb3436a..94a8bcf63 100644 --- a/tests/test_agent_claude.py +++ b/tests/test_agent_claude.py @@ -2597,7 +2597,10 @@ def test_windows_launch_preserves_prompt_as_literal_argv(self, monkeypatch, tmp_ native_binary = tmp_path / "Claude Code" / "claude.exe" prompt = 'keep "quotes" & pipes | and %PATH% literal' calls: list[list[str]] = [] - monkeypatch.setattr(claude.os, "name", "nt") + # Keep the platform simulation local so filesystem readers use the real host paths. + monkeypatch.setattr( + claude, "os", SimpleNamespace(name="nt", path=os.path, environ=os.environ) + ) monkeypatch.setattr(claude.shutil, "which", lambda _binary: str(native_binary)) monkeypatch.setattr(claude, "exec_or_spawn", lambda argv: calls.append(argv)) diff --git a/tests/test_agent_codex.py b/tests/test_agent_codex.py index 935e9712e..a4d7e2932 100644 --- a/tests/test_agent_codex.py +++ b/tests/test_agent_codex.py @@ -4,6 +4,7 @@ import json import os +import tomllib from pathlib import Path from unittest.mock import Mock @@ -993,6 +994,63 @@ def test_sets_oauth_token(self, tmp_path, monkeypatch): assert os.environ["OAUTH_TOKEN"] == "fresh-token" assert launches[0][-1] == "--search" + @pytest.mark.parametrize("version", ["0.133.0", "0.154.0"]) + def test_non_routed_launch_suppresses_legacy_plugin(self, tmp_path, monkeypatch, version): + launches = self._patch(tmp_path, monkeypatch) + monkeypatch.setattr(codex, "agent_version", lambda _binary: version) + plugin = "model-orchestrator@isaac-sync-eng-plugin-marketplace-experimental" + user_config = tmp_path / "config.toml" + user_config.write_text( + f'[plugins."{plugin}"]\nenabled = true\n' + '[plugins."unrelated@marketplace"]\nenabled = true\n' + ) + project = tmp_path / "project" + (project / ".codex").mkdir(parents=True) + project_config = project / ".codex/config.toml" + project_config.write_text( + '[plugins."model-orchestrator@project"]\nenabled = true\n' + '[plugins."unrelated@project"]\nenabled = true\n' + ) + paths = (user_config, codex.CODEX_CONFIG_PATH, project_config) + before = {path: path.read_bytes() for path in paths} + + codex.launch( + {"workspace": WS}, ["--cd", str(project), "exec", "hello"], options=LaunchOptions() + ) + + (argv,) = launches + override = next(arg for arg in argv if arg.startswith("plugins=")) + assert tomllib.loads(override) == { + "plugins": { + plugin: {"enabled": False}, + "model-orchestrator@project": {"enabled": False}, + } + } + assert argv[-4:] == ["--cd", str(project), "exec", "hello"] + assert {path: path.read_bytes() for path in paths} == before + + def test_legacy_suppression_preserves_profile_plugin_overrides(self, tmp_path, monkeypatch): + launches = self._patch(tmp_path, monkeypatch) + profile = codex.CODEX_CONFIG_PATH + profile.write_text( + profile.read_text() + '[plugins."model-orchestrator@marketplace"]\nenabled = true\n' + '[plugins."unrelated@marketplace"]\nenabled = false\n' + ) + user_config = tmp_path / "config.toml" + user_config.write_text('[plugins."unrelated@marketplace"]\nenabled = true\n') + before = user_config.read_bytes(), profile.read_bytes() + + codex.launch({"workspace": WS}, ["exec", "hello"], options=LaunchOptions()) + + (override,) = [arg for arg in launches[0] if arg.startswith("plugins=")] + assert tomllib.loads(override) == { + "plugins": { + "model-orchestrator@marketplace": {"enabled": False}, + "unrelated@marketplace": {"enabled": False}, + } + } + assert (user_config.read_bytes(), profile.read_bytes()) == before + @pytest.mark.parametrize("custom_catalog", [None, "/user/isaac-app-model-catalog.json"]) def test_native_update_detaches_catalog_without_discovery( self, tmp_path, monkeypatch, custom_catalog @@ -1486,6 +1544,9 @@ def test_catalog_cleanup_does_not_mask_write_failure(self, tmp_path, monkeypatch def test_injects_otel_config_when_tracing_enabled(self, tmp_path, monkeypatch): self._patch(tmp_path, monkeypatch) + user_config = tmp_path / "config.toml" + user_config.write_text('[plugins."model-orchestrator@marketplace"]\nenabled = true\n') + before = user_config.read_bytes() server = Mock(server_address=("127.0.0.1", 54321)) cache = Mock() client = Mock() @@ -1519,6 +1580,11 @@ def start_otel_proxy(workspace, token_provider): assert 'protocol = "binary"' in otel assert "Authorization" not in otel # no credential in argv; the proxy injects it assert argv[-2:] == ["exec", "hi"] + override = next(arg for arg in argv if arg.startswith("plugins=")) + assert tomllib.loads(override) == { + "plugins": {"model-orchestrator@marketplace": {"enabled": False}} + } + assert user_config.read_bytes() == before def test_no_otel_config_when_tracing_disabled(self, tmp_path, monkeypatch): launches = self._patch(tmp_path, monkeypatch) diff --git a/tests/test_claude_smart_routing_v2.py b/tests/test_claude_smart_routing_v2.py index 6179e3a75..15faccc7b 100644 --- a/tests/test_claude_smart_routing_v2.py +++ b/tests/test_claude_smart_routing_v2.py @@ -19,7 +19,7 @@ def _plugin_agent_models(plugin_dir: Path) -> set[str]: models = set() - for agent_path in (plugin_dir / "agents").glob("*.md"): + for agent_path in (plugin_dir / "agents").glob(f"{v2.CLAUDE_ROUTED_AGENT_PREFIX}*.md"): model_line = next( line for line in agent_path.read_text().splitlines() if line.startswith("model: ") ) @@ -482,7 +482,8 @@ def send_signal(self, _signal): assert v2.ENABLE_SMART_ROUTING_ENV_VAR not in env assert claude_hooks.FIRST_PROMPT_SOCKET_ENV not in env # Subagent routing is fully wired; only the first-prompt machinery is absent. - assert "UserPromptSubmit" not in settings["hooks"] + assert "route-first-prompt" not in str(settings["hooks"]) + assert "ucode.smart_routing.orchestrator" in str(settings["hooks"]["UserPromptSubmit"]) assert "route-subagent" in str(settings["hooks"]["PreToolUse"]) assert settings["modelOverrides"] == {"claude-opus-4-8": "system.ai.claude-opus-4-8"} assert captured["plugin_models"] == {"system.ai.claude-opus-4-8"} diff --git a/tests/test_claude_windows_smart_routing.py b/tests/test_claude_windows_smart_routing.py index b6db0f462..6a3807b25 100644 --- a/tests/test_claude_windows_smart_routing.py +++ b/tests/test_claude_windows_smart_routing.py @@ -10,12 +10,12 @@ from ucode.agents import claude from ucode.databricks import AnthropicModelCatalog -from ucode.smart_routing import session_env, v2 +from ucode.smart_routing import orchestrator, session_env, v2 def _plugin_agent_models(plugin_dir: Path) -> set[str]: models = set() - for agent_path in (plugin_dir / "agents").glob("*.md"): + for agent_path in (plugin_dir / "agents").glob(f"{v2.CLAUDE_ROUTED_AGENT_PREFIX}*.md"): model_line = next( line for line in agent_path.read_text().splitlines() if line.startswith("model: ") ) @@ -43,6 +43,8 @@ def test_windows_subagent_routing_uses_native_binary_without_unix_imports(tmp_pa warnings: list[str] = [] host_os_name = v2.os.name + skill_directory = orchestrator.skill_directory() + monkeypatch.setattr(orchestrator, "skill_directory", lambda: skill_directory) path_type = type(tmp_path) monkeypatch.setattr(v2.os, "name", "nt") # ``pathlib.Path`` follows the process-wide os.name even on this Linux test host. Keep the @@ -118,7 +120,8 @@ def send_signal(self, _signal): assert captured["plugin_models"] == {"system.ai.claude-opus-4-8"} assert settings["env"][v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR] == "1" assert v2.ENABLE_SMART_ROUTING_ENV_VAR not in settings["env"] - assert "UserPromptSubmit" not in settings["hooks"] + assert "route-first-prompt" not in str(settings["hooks"]) + assert "ucode.smart_routing.orchestrator" in str(settings["hooks"]["UserPromptSubmit"]) assert "route-subagent" in str(settings["hooks"]["PreToolUse"]) assert settings["modelOverrides"] == {"claude-opus-4-8": "system.ai.claude-opus-4-8"} assert not captured["settings_path"].exists() diff --git a/tests/test_cli.py b/tests/test_cli.py index 7e5124b8b..df4592e1e 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -677,6 +677,34 @@ def test_codex_and_claude_share_smart_routing_policy( assert options.launch_smart_routing is expected + @pytest.mark.parametrize("tool", ["claude", "codex"]) + @pytest.mark.parametrize("routing_enabled", [False, True]) + def test_launch_discards_parent_session_before_applying_eligibility( + self, monkeypatch, tmp_path, tool, routing_enabled + ): + from ucode.smart_routing import session_env + + parent = tmp_path / "parent-env.json" + parent.write_text("{}") + monkeypatch.setenv(session_env.SESSION_ENV_VAR, str(parent)) + monkeypatch.setenv(session_env.SESSION_PYTHON_ENV_VAR, "/parent/python") + monkeypatch.setenv("ENABLE_SMART_ROUTING_V2", "1" if routing_enabled else "0") + observed = [] + + def launch(_tool, _state, _args, *, options): + observed.append( + (options.launch_smart_routing, os.environ.get(session_env.SESSION_ENV_VAR)) + ) + + with _launch_policy_patches(None) as calls: + calls["launch"].side_effect = launch + result = runner.invoke(app, [tool]) + + assert result.exit_code == 0, result.output + assert observed == [(routing_enabled, None)] + assert os.environ[session_env.SESSION_ENV_VAR] == str(parent) + assert parent.read_text() == "{}" + @pytest.mark.parametrize( ("tool_args", "expected"), [ diff --git a/tests/test_codex_config.py b/tests/test_codex_config.py index c1fe6c5b0..219e626f4 100644 --- a/tests/test_codex_config.py +++ b/tests/test_codex_config.py @@ -1,7 +1,14 @@ from __future__ import annotations +import json +import subprocess +import sys +import textwrap + +import pytest import tomlkit +from ucode import codex_config from ucode.agents import codex from ucode.codex_config import codex_config_args @@ -60,3 +67,150 @@ def test_renders_nested_tables_from_parsed_profile(self): assert 'http_headers = {User-Agent = "ucode"}' in provider_override assert 'auth = {command = "ucode", args = ["codex-token"]}' in provider_override assert 'tui={model_availability_nux = {"gpt-5.6-sol" = 1}}' in args + + def test_renders_native_hook_defaults_as_optional_fields(self): + config = { + "hooks": { + "UserPromptSubmit": [ + { + "matcher": None, + "hooks": [ + {"type": "command", "command": "policy", "command_windows": None} + ], + } + ] + } + } + + args = codex_config_args(config) + + assert tomlkit.parse(args[1]) == { + "hooks": { + "UserPromptSubmit": [ + { + "hooks": [{"type": "command", "command": "policy"}], + } + ] + } + } + + +@pytest.mark.parametrize( + "args", + [ + ["--cd", "project"], + ["--cd=project"], + ["-C", "project"], + ["-Cproject"], + ["-C=project"], + ["--cd", "wrong", "--cd", "project", "--", "--cd", "ignored"], + ], +) +def test_codex_working_directory(tmp_path, monkeypatch, args): + monkeypatch.chdir(tmp_path) + assert codex_config.codex_working_directory(args) == tmp_path / "project" + + +def test_native_lookup_preserves_caller_config_overrides(): + overrides = [ + "-c", + 'projects={"/project"={trust_level="untrusted"}}', + "--config", + 'model="example"', + "--config=features.hooks=true", + "-cfeatures.search=false", + "--disable", + "remote_control", + ] + args = ["--cd", "/project", *overrides, "--", "--config=prompt-text"] + + assert codex_config.codex_cli_config_args(args) == overrides + assert args[-1] == "--config=prompt-text" + + +class TestReadEffectiveCodexConfig: + def _server(self, monkeypatch, script): + processes = [] + calls = [] + + def start(argv, **kwargs): + # Substitute only the native protocol peer; exercise real pipes and cleanup. + calls.append((argv, kwargs)) + process = subprocess.Popen([sys.executable, "-c", textwrap.dedent(script)], **kwargs) + processes.append(process) + return process + + monkeypatch.setattr(codex_config.subprocess_cross_os, "popen", start) + return processes, calls + + def test_reads_native_result_after_handshake(self, tmp_path, monkeypatch, capsys): + config = {"hooks": {"UserPromptSubmit": [{"hooks": [{"command": "project-policy"}]}]}} + script = """ + import json, sys + initial = json.loads(sys.stdin.readline()) + assert initial['method'] == 'initialize' + print(json.dumps({'id': initial['id'], 'result': {}}), flush=True) + assert json.loads(sys.stdin.readline())['method'] == 'initialized' + request = json.loads(sys.stdin.readline()) + assert request['method'] == 'config/read' + assert request['params'] == {'cwd': CWD, 'includeLayers': False} + print(json.dumps({'method': 'notification'}), flush=True) + print(json.dumps({'id': request['id'], 'result': {'config': CONFIG}}), flush=True) + assert sys.stdin.read() == '' + """.replace("CWD", repr(str(tmp_path))).replace("CONFIG", repr(config)) + processes, calls = self._server(monkeypatch, script) + + assert ( + codex_config.read_effective_codex_config( + "/selected/codex", cwd=tmp_path, config_args=["--config", 'model="example"'] + ) + == config + ) + + assert calls[0][0] == [ + "/selected/codex", + "app-server", + "--config", + 'model="example"', + "--listen", + "stdio://", + ] + assert calls[0][1]["cwd"] == tmp_path + assert processes[0].returncode == 0 + assert processes[0].stdin.closed and processes[0].stdout.closed + assert capsys.readouterr().out == "" + + @pytest.mark.parametrize( + "response", + [ + "not json", + json.dumps({"id": 1, "error": {"message": "invalid config"}}), + json.dumps({"id": 1, "result": None}), + ], + ) + def test_protocol_errors_do_not_fall_back_to_incomplete_hooks( + self, tmp_path, monkeypatch, response + ): + processes, _ = self._server( + monkeypatch, + f""" + import sys + sys.stdin.readline() + print({response!r}, flush=True) + """, + ) + + with pytest.raises(RuntimeError, match="--disable-smart-routing"): + codex_config.read_effective_codex_config("codex", cwd=tmp_path, config_args=[]) + + assert processes[0].poll() is not None + + def test_unresponsive_server_is_reaped(self, tmp_path, monkeypatch): + monkeypatch.setattr(codex_config, "CONFIG_READ_TIMEOUT_SECONDS", 0.05) + processes, _ = self._server(monkeypatch, "import time; time.sleep(60)") + + with pytest.raises(RuntimeError, match="Could not read Codex configuration"): + codex_config.read_effective_codex_config("codex", cwd=tmp_path, config_args=[]) + + assert processes[0].poll() is not None + assert processes[0].stdin.closed and processes[0].stdout.closed diff --git a/tests/test_codex_smart_routing_v2.py b/tests/test_codex_smart_routing_v2.py index 287e02516..d999d3638 100644 --- a/tests/test_codex_smart_routing_v2.py +++ b/tests/test_codex_smart_routing_v2.py @@ -2,6 +2,8 @@ import json import os +import tomllib +from copy import deepcopy from types import SimpleNamespace import pytest @@ -13,6 +15,14 @@ WS = "https://example.databricks.com" +@pytest.fixture(autouse=True) +def native_config(monkeypatch): + # Launch tests isolate the native Codex process; its protocol is tested separately. + config = {} + monkeypatch.setattr(v2, "read_effective_codex_config", lambda *args, **kwargs: config) + return config + + def test_smart_routing_switch_message_is_boxed(): message = v2.format_routing_notice("model-x", "Because X.") @@ -180,15 +190,26 @@ def test_startup_config_precedence( @pytest.mark.parametrize( ("platform_name", "tui_has_provider"), [("posix", False), ("nt", True)] ) + @pytest.mark.parametrize("legacy_plugin", [False, True]) def test_owns_app_server_interposer_and_tui_lifecycle( - self, monkeypatch, platform_name, tui_has_provider + self, tmp_path, monkeypatch, platform_name, tui_has_provider, legacy_plugin ): processes = [] interposer_args = {} stopped = [] token_calls = [] monkeypatch.setenv(v2.ENABLE_SMART_ROUTING_ENV_VAR, "1") - monkeypatch.setenv("CODEX_HOME", "/user/codex-home") + monkeypatch.setenv("CODEX_HOME", str(tmp_path)) + user_config = tmp_path / "config.toml" + user_config.write_text( + '[plugins."unrelated@marketplace"]\nenabled = true\n' + + ( + '[plugins."model-orchestrator@marketplace"]\nenabled = true\n' + if legacy_plugin + else "" + ) + ) + before = user_config.read_bytes() monkeypatch.setattr(v2, "os", SimpleNamespace(name=platform_name, environ=os.environ)) monkeypatch.setattr(codex, "ug_version", lambda: "0.1.0") monkeypatch.setattr(codex, "agent_version", lambda binary: "0.148.0") @@ -263,26 +284,37 @@ def start_interposer(*args, **kwargs): assert "--profile myprof" in hook_override assert "--model system.ai.gpt-5-6-sol" in hook_override assert "--model system.ai.glm-5-2" in hook_override - assert processes[0].argv[10:12] == [ - "--config", - ( - "shell_environment_policy.set.UCODE_SESSION_ENV_FILE=" - f'"{os.environ["UCODE_SESSION_ENV_FILE"]}"' - ), - ] - assert processes[0].argv[12:14] == [ - "--config", + config_values = processes[0].argv[3:-2:2] + assert ( + "shell_environment_policy.set.UCODE_SESSION_ENV_FILE=" + f'"{os.environ["UCODE_SESSION_ENV_FILE"]}"' + ) in config_values + assert ( "shell_environment_policy.set.UCODE_SMART_ROUTER_PYTHON=" - + json.dumps(os.environ["UCODE_SMART_ROUTER_PYTHON"]), - ] - assert processes[0].argv[14:] == [ + + json.dumps(os.environ["UCODE_SMART_ROUTER_PYTHON"]) + ) in config_values + assert 'shell_environment_policy.set.ENABLE_SMART_ROUTING_V2="1"' in config_values + assert "features.hooks=true" in config_values + plugin_overrides = [value for value in config_values if value.startswith("plugins=")] + if legacy_plugin: + (override,) = plugin_overrides + assert tomllib.loads(override) == { + "plugins": {"model-orchestrator@marketplace": {"enabled": False}} + } + else: + assert plugin_overrides == [] + for event in ("UserPromptSubmit", "SessionStart"): + hook = next(value for value in config_values if value.startswith(f"hooks.{event}=")) + assert "ucode.smart_routing.orchestrator" in hook + assert processes[0].argv[-2:] == [ "--listen", "ws://127.0.0.1:41001", ] assert processes[0].kwargs["env"][v2.OAUTH_TOKEN_ENV_VAR] == "token-1" - assert processes[0].kwargs["env"]["CODEX_HOME"] == "/user/codex-home" + assert processes[0].kwargs["env"]["CODEX_HOME"] == str(tmp_path) tui_argv = processes[1].argv expected_tui_args = [ + *(["--config", plugin_overrides[0]] if legacy_plugin else []), "--remote", "ws://127.0.0.1:41002", "--model", @@ -313,6 +345,7 @@ def start_interposer(*args, **kwargs): assert interposer_args["kwargs"]["switch_message_fn"] is v2.format_routing_notice assert stopped == [True] assert processes[0].terminated is True + assert user_config.read_bytes() == before def test_managed_http_headers_reach_app_server_config(self, monkeypatch): # Smart routing rebuilds the overlay and passes it to the app-server as `-c` overrides that @@ -367,16 +400,22 @@ def send_signal(self, _signal): assert "x-databricks-workspace" in provider_arg assert "eng-ml-inference" in provider_arg - def test_subagent_only_launch_runs_tui_directly(self, tmp_path, monkeypatch): + @pytest.mark.parametrize("legacy_plugin", [False, True]) + def test_subagent_only_launch_runs_tui_directly(self, tmp_path, monkeypatch, legacy_plugin): monkeypatch.setenv(v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR, "1") monkeypatch.setenv("CODEX_HOME", str(tmp_path)) + user_config = tmp_path / "config.toml" + user_config.write_text( + '[plugins."model-orchestrator@marketplace"]\nenabled = true\n' if legacy_plugin else "" + ) + before = user_config.read_bytes() monkeypatch.setattr(codex, "ug_version", lambda: "0.1.0") monkeypatch.setattr(codex, "agent_version", lambda binary: "0.148.0") monkeypatch.setattr(v2, "get_databricks_token", lambda *_args, **_kwargs: "token") monkeypatch.setattr( v2.subprocess, "Popen", - lambda *_args, **_kwargs: pytest.fail("subagent-only routing spawns no app-server"), + lambda *_args, **_kwargs: pytest.fail("subagent-only routing starts no routing server"), ) monkeypatch.setattr( codex_interposer, @@ -416,6 +455,21 @@ def fake_exec(argv): "shell_environment_policy.set.UCODE_SMART_ROUTER_PYTHON=" + json.dumps(os.environ["UCODE_SMART_ROUTER_PYTHON"]) ) in argv + assert 'shell_environment_policy.set.ENABLE_SMART_ROUTING_SUBAGENT_ONLY="1"' in argv + assert "features.hooks=true" in argv + plugin_overrides = [arg for arg in argv if arg.startswith("plugins=")] + if legacy_plugin: + (override,) = plugin_overrides + assert tomllib.loads(override) == { + "plugins": {"model-orchestrator@marketplace": {"enabled": False}} + } + else: + assert plugin_overrides == [] + assert user_config.read_bytes() == before + assert any( + arg.startswith("hooks.UserPromptSubmit=") and "ucode.smart_routing.orchestrator" in arg + for arg in argv + ) # The hook subprocesses inherit the launch environment and pass the routing gate. assert os.environ[v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR] == "1" assert os.environ[v2.OAUTH_TOKEN_ENV_VAR] == "token" @@ -433,15 +487,77 @@ def test_v2_pre_tool_hook_preserves_user_hooks(self, tmp_path, monkeypatch): ) monkeypatch.setenv("CODEX_HOME", str(codex_home)) - configured = v2._v2_pre_tool_use_hooks( + configured = v2._v2_hooks( {"workspace": WS, "profile": "myprof"}, ["system.ai.gpt-5-6-sol"], - ) + tomllib.loads((codex_home / "config.toml").read_text()), + )["PreToolUse"] assert configured[0]["hooks"][0]["command"] == "user-policy" assert configured[1]["matcher"] == "Agent|.*spawn_agent$" assert "--model system.ai.gpt-5-6-sol" in configured[1]["hooks"][0]["command"] + def test_subagent_launch_composes_only_native_resolved_events(self, tmp_path, monkeypatch): + monkeypatch.setenv(v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR, "1") + monkeypatch.setattr(v2, "get_databricks_token", lambda *args: "token") + project = tmp_path / "project" + (project / ".codex").mkdir(parents=True) + project_config = project / ".codex/config.toml" + project_config.write_text('[plugins."model-orchestrator@project"]\nenabled = true\n') + before = project_config.read_bytes() + effective = { + "hooks": { + event: [{"hooks": [{"type": "command", "command": f"native-{event}"}]}] + for event in ("PreToolUse", "UserPromptSubmit", "SessionStart", "Stop") + } + } + saved = deepcopy(effective) + reads, launches = [], [] + caller_config = ["--config", 'projects={"/other"={trust_level="untrusted"}}'] + + def native_read(binary, **kwargs): + reads.append((binary, kwargs)) + return effective + + def launch(argv): + launches.append(argv) + raise SystemExit(0) + + monkeypatch.setattr(v2, "read_effective_codex_config", native_read) + monkeypatch.setattr(v2, "exec_or_spawn", launch) + with pytest.raises(SystemExit): + v2.launch_codex( + {"workspace": WS, "codex_models": ["gpt-5.6-sol"]}, + [*caller_config, "--cd", str(project)], + binary="/selected/codex", + start_model="gpt-5.6-sol", + render_overlay=lambda *args, **kwargs: {"model": "gpt-5.6-sol"}, + ) + + assert reads == [ + ( + "/selected/codex", + { + "cwd": project, + "config_args": ["--config", 'model="gpt-5.6-sol"', *caller_config], + }, + ) + ] + (argv,) = launches + for event in ("PreToolUse", "UserPromptSubmit", "SessionStart"): + value = next(arg for arg in argv if arg.startswith(f"hooks.{event}=")) + groups = tomllib.loads(value)["hooks"][event] + assert groups[0] == saved["hooks"][event][0] + assert len(groups) == 2 + assert not any(arg.startswith("hooks.Stop=") for arg in argv) + plugin_arg = next(arg for arg in argv if arg.startswith("plugins=")) + assert tomllib.loads(plugin_arg) == { + "plugins": {"model-orchestrator@project": {"enabled": False}} + } + assert argv[-2:] == ["--cd", str(project)] + assert effective == saved + assert project_config.read_bytes() == before + def test_v2_pre_tool_hook_replaces_existing_ucode_hook(self, tmp_path, monkeypatch): monkeypatch.setattr("ucode.databricks.ug_binary", lambda: "/bin/ug") codex_home = tmp_path / ".codex" @@ -456,10 +572,11 @@ def test_v2_pre_tool_hook_replaces_existing_ucode_hook(self, tmp_path, monkeypat ) monkeypatch.setenv("CODEX_HOME", str(codex_home)) - configured = v2._v2_pre_tool_use_hooks( + configured = v2._v2_hooks( {"workspace": WS, "profile": "myprof"}, ["system.ai.gpt-5-6-sol"], - ) + tomllib.loads((codex_home / "config.toml").read_text()), + )["PreToolUse"] routing_commands = [ hook["command"] diff --git a/tests/test_integration_evidence.py b/tests/test_integration_evidence.py index 910984916..cf04efc76 100644 --- a/tests/test_integration_evidence.py +++ b/tests/test_integration_evidence.py @@ -10,6 +10,7 @@ SubagentCalculation, assert_no_terminal_api_error, assistant_answer_contains, + tool_outputs, ) @@ -216,3 +217,76 @@ def test_completed_task_model_assertion_requires_exact_singleton(tmp_path, agent session = _transcript_session(tmp_path, agent, {"parent.jsonl": records}) with pytest.raises(AssertionError): evidence.assert_completed_task_model(session, agent, "value", "expected") + + +@pytest.mark.parametrize( + "agent,record", + [ + ( + "claude", + { + "type": "user", + "message": {"content": [{"type": "tool_result", "content": "confirmed"}]}, + }, + ), + ( + "claude", + { + "type": "user", + "message": { + "content": [ + { + "type": "tool_result", + "content": [{"type": "text", "text": "confirmed"}], + } + ] + }, + }, + ), + ( + "codex", + { + "type": "response_item", + "payload": {"type": "function_call_output", "output": "confirmed"}, + }, + ), + ( + "codex", + { + "type": "response_item", + "payload": { + "type": "custom_tool_call_output", + "output": [{"type": "input_text", "text": "confirmed"}], + }, + }, + ), + ], +) +def test_tool_outputs_read_native_string_and_block_results(agent, record): + assert tool_outputs(agent, [record]) == ["confirmed"] + + +@pytest.mark.parametrize("agent", ["claude", "codex"]) +def test_tool_outputs_exclude_user_echoes_and_assistant_claims(agent): + records = [ + {"type": "user", "message": {"role": "user", "content": "confirmed"}}, + { + "type": "assistant", + "message": {"role": "assistant", "content": [{"type": "text", "text": "confirmed"}]}, + }, + { + "type": "response_item", + "payload": { + "type": "message", + "role": "assistant", + "content": [{"type": "output_text", "text": "confirmed"}], + }, + }, + { + "type": "user", + "message": { + "content": [{"type": "tool_result", "content": "confirmed", "is_error": True}] + }, + }, + ] + assert tool_outputs(agent, records) == [] diff --git a/tests/test_managed_files.py b/tests/test_managed_files.py index 606a7ba8f..6034a53ea 100644 --- a/tests/test_managed_files.py +++ b/tests/test_managed_files.py @@ -17,6 +17,7 @@ import ucode.codex_config as codex_config import ucode.config_io as config_io from ucode import managed_files +from ucode.codex_config import codex_managed_config_path _REAL_SUDO_REPLACE = managed_files._sudo_replace @@ -506,7 +507,8 @@ def test_real_managed_path_helpers_are_allowlisted(self, os_enum, monkeypatch): monkeypatch.setattr(codex_config, "current_os", lambda: os_enum) allowed = managed_files._SUDO_REPLACE_TARGETS[os_enum] assert claude_agent._managed_settings_path() in allowed - assert codex_config.codex_managed_config_path() in allowed + # Exercise the real helper, captured before launch fixtures isolate machine config. + assert codex_managed_config_path() in allowed @pytest.mark.skipif(sys.platform == "win32", reason="The managed writer is Unix-only") diff --git a/tests/test_orchestrator.py b/tests/test_orchestrator.py new file mode 100644 index 000000000..148827983 --- /dev/null +++ b/tests/test_orchestrator.py @@ -0,0 +1,233 @@ +"""Orchestration follows the same launch eligibility and live state as routing.""" + +import copy +import io +import json +import os +import shlex +import subprocess +import sys +from pathlib import Path + +import pytest +from typer.testing import CliRunner + +from ucode import cli, skills +from ucode.smart_routing import orchestrator, routing, session_env, v2 + + +@pytest.fixture +def routed_session(monkeypatch): + monkeypatch.setenv(v2.ENABLE_SMART_ROUTING_ENV_VAR, "1") + monkeypatch.delenv(v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR, raising=False) + monkeypatch.delenv("ISAAC_LAUNCH_MODE", raising=False) + monkeypatch.setattr(skills, "_skills_source", lambda: Path(__file__).parents[1] / "skills") + return session_env.start_session() + + +@pytest.mark.parametrize("agent", ["claude", "codex"]) +def test_toggle_supersedes_workflow_and_reenables_it(routed_session, agent): + payload = {"hook_event_name": "UserPromptSubmit"} + workflow = orchestrator.hook_output(payload)["hookSpecificOutput"]["additionalContext"] + assert "Apply the UG model-orchestrator workflow" in workflow + runner = CliRunner() + + off = runner.invoke(cli.app, [agent, "--disable-smart-routing"]) + assert off.exit_code == 0, off.output + assert "do not start new automatic" in " ".join(off.output.split()) + assert not v2.smart_routing_enabled(session_env.effective_environment()) + assert not orchestrator.enabled() + assert orchestrator.hook_output(payload)["hookSpecificOutput"]["additionalContext"] == ( + orchestrator.DISABLED_CONTEXT + ) + with pytest.raises(ValueError, match="supersedes any earlier"): + orchestrator.require_enabled() + + on = runner.invoke(cli.app, [agent, "--enable-smart-routing"]) + assert on.exit_code == 0, on.output + assert "Automatic orchestration is on" in on.output + assert v2.smart_routing_enabled(session_env.effective_environment()) + assert orchestrator.hook_output(payload)["hookSpecificOutput"]["additionalContext"] == workflow + + +@pytest.mark.parametrize("mode", ["no-session", "missing-session", "off", "omni"]) +def test_retained_skill_cannot_enable_orchestration(routed_session, monkeypatch, mode): + if mode == "no-session": + monkeypatch.delenv(session_env.SESSION_ENV_VAR) + elif mode == "missing-session": + routed_session.unlink() + elif mode == "off": + session_env.set_session_environment(dict.fromkeys(v2.SMART_ROUTING_ENV_KEYS, "0")) + else: + monkeypatch.setenv("ISAAC_LAUNCH_MODE", " OMNI ") + assert (orchestrator.skill_directory() / "SKILL.md").is_file() + assert not orchestrator.enabled() + with pytest.raises(ValueError, match="do not start new automatic delegation"): + orchestrator.require_enabled() + + +@pytest.mark.parametrize( + "flags", + [ + {v2.ENABLE_SMART_ROUTING_ENV_VAR: "1"}, + {v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1"}, + {v2.ENABLE_SMART_ROUTING_ENV_VAR: "0", v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1"}, + {v2.ENABLE_SMART_ROUTING_ENV_VAR: "0"}, + {}, + ], +) +def test_full_and_subagent_routing_share_the_gate(routed_session, flags): + env = {session_env.SESSION_ENV_VAR: str(routed_session), **flags} + assert orchestrator.enabled(env) == v2.smart_routing_enabled( + session_env.effective_environment(env) + ) + + +def test_nested_launch_does_not_inherit_eligibility(routed_session): + before = os.environ[session_env.SESSION_ENV_VAR] + with session_env.fresh_launch(): + assert session_env.SESSION_PYTHON_ENV_VAR not in os.environ + assert not orchestrator.enabled() + # A newly eligible launch creates its own session instead of sharing the parent's toggle. + child_session = session_env.start_session() + assert child_session != routed_session + assert orchestrator.enabled() + session_env.set_session_environment(dict.fromkeys(v2.SMART_ROUTING_ENV_KEYS, "0")) + assert os.environ[session_env.SESSION_ENV_VAR] == before + assert orchestrator.enabled() + + +@pytest.mark.parametrize( + "payload,active", + [ + ({"hook_event_name": "UserPromptSubmit"}, True), + ({"hook_event_name": "UserPromptSubmit", "agent_type": "custom-root"}, True), + ({"hook_event_name": "SessionStart", "source": "compact"}, True), + ({"hook_event_name": "SessionStart", "source": "startup"}, False), + ({"hook_event_name": "UserPromptSubmit", "agent_id": "child"}, False), + ({"hook_event_name": "SessionStart", "source": "compact", "agent_id": "child"}, False), + ({"hook_event_name": []}, False), + ({"hook_event_name": "Unknown"}, False), + ([], False), + (None, False), + ], +) +def test_only_root_prompt_and_compaction_load_workflow(routed_session, payload, active): + output = orchestrator.hook_output(payload) + if not active: + assert output is None + return + context = output["hookSpecificOutput"]["additionalContext"] + directory = orchestrator.skill_directory() + assert str(directory) in context + assert context.endswith((directory / "SKILL.md").read_text()) + + +@pytest.mark.parametrize("payload", ["", "{", "null", "[]", '"text"', "{}"]) +def test_hook_entry_point_ignores_invalid_payload(monkeypatch, capsys, payload): + monkeypatch.setattr(sys, "stdin", io.StringIO(payload)) + orchestrator.main() + assert capsys.readouterr().out == "" + + +@pytest.mark.parametrize("state", ["missing", "directory", "invalid-encoding"]) +def test_unreadable_workflow_is_not_injected(routed_session, tmp_path, monkeypatch, state): + monkeypatch.setattr(orchestrator, "skill_directory", lambda: tmp_path) + skill = tmp_path / "SKILL.md" + if state == "directory": + skill.mkdir() + elif state == "invalid-encoding": + skill.write_bytes(b"\xff") + assert orchestrator.hook_output({"hook_event_name": "UserPromptSubmit"}) is None + + +def test_hooks_preserve_user_handlers_and_replace_only_ug_handlers(monkeypatch): + monkeypatch.setattr(sys, "executable", "/launch installation/bin/python") + user = {"hooks": [{"type": "command", "command": "user-policy"}]} + doc = { + "hooks": { + "UserPromptSubmit": [copy.deepcopy(user)], + "SessionStart": [copy.deepcopy(user)], + "PreToolUse": [copy.deepcopy(user)], + } + } + orchestrator.sync_hooks(doc, agent="codex") + first = copy.deepcopy(doc) + orchestrator.sync_hooks(doc, agent="codex") + assert doc == first + assert doc["hooks"]["PreToolUse"] == [user] + for event in ("UserPromptSubmit", "SessionStart"): + groups = doc["hooks"][event] + assert len(groups) == 2 + assert groups[0] == user + hook = groups[1]["hooks"][0] + assert shlex.split(hook["command"]) == [sys.executable, "-m", orchestrator.HOOK_MODULE] + assert hook["command_windows"] == subprocess.list2cmdline( + [sys.executable, "-m", orchestrator.HOOK_MODULE] + ) + assert doc["hooks"]["SessionStart"][1]["matcher"] == "compact" + + +def test_codex_launch_merges_prompt_and_compaction_hooks(tmp_path, monkeypatch): + config = tmp_path / "config.toml" + config.write_text( + '[[hooks.UserPromptSubmit]]\n[[hooks.UserPromptSubmit.hooks]]\ncommand = "user-prompt"\n' + '[[hooks.SessionStart]]\nmatcher = "compact"\n' + '[[hooks.SessionStart.hooks]]\ncommand = "user-compact"\n' + ) + before = config.read_bytes() + monkeypatch.setenv("CODEX_HOME", str(tmp_path)) + hooks = v2._v2_hooks( + {"workspace": "https://example.com"}, ["gpt-6-sol"], v2.config_io.read_toml_safe(config) + ) + assert hooks["UserPromptSubmit"][0]["hooks"][0]["command"] == "user-prompt" + assert hooks["SessionStart"][0]["hooks"][0]["command"] == "user-compact" + assert orchestrator.HOOK_MODULE in hooks["UserPromptSubmit"][1]["hooks"][0]["command"] + assert config.read_bytes() == before + + +def test_routed_claude_plugin_contains_unchanged_roles(routed_session, tmp_path): + v2._write_routed_claude_plugin(tmp_path, ["system.ai.claude-sonnet-4-6"]) + manifest = json.loads((tmp_path / ".claude-plugin/plugin.json").read_text()) + assert manifest["name"] == "ug-smart-router" + templates = list((orchestrator.skill_directory() / "agents").glob("*.md")) + assert {path.stem for path in templates} == { + "explorer", + "researcher", + "worker", + "tester", + "reviewer", + } + for template in templates: + assert (tmp_path / "agents" / template.name).read_bytes() == template.read_bytes() + + +def test_role_contract_survives_claude_model_routing(monkeypatch): + monkeypatch.setattr( + v2, + "_request_claude_routing_decision", + lambda *_args: ( + routing.RoutingDecision( + model="system.ai.claude-sonnet-4-6", raw_model="claude-sonnet-4-6" + ), + None, + ), + ) + contract = ( + "Act as the researcher. Verify the API using the connected documentation tool. " + "Do not edit files or delegate. Stop on auth failure. Return source links and evidence." + ) + payload = { + "tool_name": "Agent", + "tool_input": {"subagent_type": "ug-smart-router:researcher", "prompt": contract}, + } + result = v2.route_claude_pre_tool_use( + payload, + workspace="https://example.com", + token="token", + available_models=["system.ai.claude-sonnet-4-6"], + ) + updated = result["hookSpecificOutput"]["updatedInput"] + assert updated["prompt"] == contract + assert updated["subagent_type"].startswith("ug-smart-router:ucode-route-") + assert "model" not in updated diff --git a/tests/test_orchestrator_config.py b/tests/test_orchestrator_config.py new file mode 100644 index 000000000..2f02ff059 --- /dev/null +++ b/tests/test_orchestrator_config.py @@ -0,0 +1,954 @@ +"""Exercise real CLI configuration, ownership, and removal without model calls.""" + +import errno +import importlib.util +import json +import os +import shutil +import stat +import subprocess +import sys +import time +from pathlib import Path +from typing import Any + +import pytest + +from ucode.os_compatibility.file_lock_cross_os import acquire_exclusive_file_lock, release_file_lock +from ucode.smart_routing import session_env + +# Only replace the external machine config path; run the real helper and file operations. +CONFIG_CLI = """ +import runpy +import sys +from pathlib import Path +from ucode import codex_config +script, managed, *arguments = sys.argv[1:] +codex_config.codex_managed_config_path = lambda: Path(managed) +sys.argv = [script, *arguments] +runpy.run_path(script, run_name="__main__") +""" + + +class ConfigHarness: + def __init__(self, root: Path, script: Path): + self.root = root + self.script = script + self.project = root / "project with spaces" + self.project.mkdir() + self.env = dict( + os.environ, + XDG_CONFIG_HOME=str(root / "config"), + CLAUDE_CONFIG_DIR=str(root / "claude"), + CODEX_HOME=str(root / "codex-home"), + ) + self.env.pop("CLAUDE_CODE_SUBAGENT_MODEL_FORCE", None) + self.env["ENABLE_SMART_ROUTING_V2"] = "1" + self.env.pop("ENABLE_SMART_ROUTING_SUBAGENT_ONLY", None) + session_env.start_session(self.env) + self.env.pop("ISAAC_LAUNCH_MODE", None) + + @property + def config(self): + return self.project / ".model-orchestrator.json" + + @property + def journal(self): + return self.config.with_name(self.config.name + ".transaction.json") + + @property + def user_config(self): + return Path(self.env["XDG_CONFIG_HOME"]) / "model-orchestrator/config.json" + + def cli(self, *args, user=False, ok=True) -> Any: + scope = ["--user"] if user else ["--project", str(self.project)] + result = subprocess.run( + [ + sys.executable, + "-c", + CONFIG_CLI, + str(self.script), + str(self.root / "managed.toml"), + *args, + *scope, + ], + env=self.env, + capture_output=True, + text=True, + timeout=20, + ) + assert (result.returncode == 0) == ok, result.stderr + return json.loads(result.stdout) if ok else result.stderr + + def set_model(self, harness="claude", role="worker", model="provider/custom-model", **kwargs): + return self.cli("set", "--harness", harness, "--role", role, "--model", model, **kwargs) + + def set_catalog(self, *models): + catalog = self.root / "codex-model-catalog.json" + catalog.write_text(json.dumps({"models": [{"slug": model} for model in models]})) + codex_home = Path(self.env["CODEX_HOME"]) + codex_home.mkdir(exist_ok=True) + (codex_home / "ucode.config.toml").write_text( + f"model_catalog_json = {json.dumps(str(catalog))}\n" + ) + return catalog + + def agent(self, role="worker", user=False): + root = Path(self.env["CLAUDE_CONFIG_DIR"]) if user else self.project / ".claude" + return ( + root / "agents" / f"model-orchestrator-custom-{'user' if user else 'project'}-{role}.md" + ) + + def interrupt(self, target, *arguments): + program = """ +import importlib.util +import os +from pathlib import Path +import sys + +script, target, *arguments = sys.argv[1:] +spec = importlib.util.spec_from_file_location("orchestrator_configure", script) +module = importlib.util.module_from_spec(spec) +spec.loader.exec_module(module) +original_replace = os.replace +original_unlink = os.unlink + +def replace(source, destination): + original_replace(source, destination) + if Path(destination) == Path(target): + os._exit(17) + +def unlink(destination, *args, **kwargs): + original_unlink(destination, *args, **kwargs) + if Path(destination) == Path(target): + os._exit(17) + +os.replace = replace +os.unlink = unlink +sys.argv = [script, *arguments] +module.main() +""" + result = subprocess.run( + [ + sys.executable, + "-c", + program, + str(self.script), + str(target), + *arguments, + "--project", + str(self.project), + ], + env=self.env, + capture_output=True, + text=True, + timeout=20, + ) + assert result.returncode == 17, result.stderr + + +@pytest.fixture +def config(tmp_path): + return ConfigHarness( + tmp_path, Path(__file__).parents[1] / "skills/orchestrate/scripts/configure.py" + ) + + +@pytest.fixture +def configure_module(config, monkeypatch): + spec = importlib.util.spec_from_file_location("orchestrator_configure", config.script) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + monkeypatch.setattr(module, "codex_managed_config_path", lambda: config.root / "managed.toml") + for key in (session_env.SESSION_ENV_VAR, "ENABLE_SMART_ROUTING_V2"): + monkeypatch.setenv(key, config.env[key]) + return module + + +def test_defaults_need_no_files(config): + claude = config.cli("show", "--harness", "claude") + codex = config.cli("show", "--harness", "codex") + assert {choice["model"] for choice in claude.values()} == {"sonnet"} + assert claude["reviewer"]["subagent_type"] == "ug-smart-router:reviewer" + assert codex["worker"] == { + "model": "gpt-5.6-luna", + "reasoning_effort": "max", + "allow_inherited_fallback": True, + } + assert list(config.project.iterdir()) == [] + assert not Path(config.env["XDG_CONFIG_HOME"]).exists() + + +@pytest.mark.parametrize("harness", ["claude", "codex"]) +@pytest.mark.parametrize("configured", [False, True]) +@pytest.mark.parametrize("state", ["off", "no-session"]) +def test_show_never_authorizes_delegation_when_routing_is_off(config, harness, configured, state): + if configured: + config.set_model(harness=harness) + if state == "off": + Path(config.env[session_env.SESSION_ENV_VAR]).write_text( + '{"ENABLE_SMART_ROUTING_V2":"0","ENABLE_SMART_ROUTING_SUBAGENT_ONLY":"0"}' + ) + else: + config.env.pop(session_env.SESSION_ENV_VAR) + error = config.cli("show", "--harness", harness, ok=False) + assert "do not start new automatic delegation" in error + # Preferences can still be managed without enabling either feature. + config.set_model(harness=harness) + config.cli("unconfigure") + + +def test_codex_managed_catalog_precedes_ug_and_user_catalogs(config): + catalog = config.set_catalog("gpt-5.6-luna") + managed_catalog = config.root / "managed-models.json" + managed_catalog.write_text('{"models":[{"slug":"system.ai.gpt-5-6-luna"}]}') + (config.root / "managed.toml").write_text( + f"model_catalog_json = {json.dumps(str(managed_catalog))}\n" + ) + (Path(config.env["CODEX_HOME"]) / "config.toml").write_text( + f"model_catalog_json = {json.dumps(str(catalog))}\n" + ) + assert config.cli("show", "--harness", "codex")["worker"]["model"] == ("system.ai.gpt-5-6-luna") + + +@pytest.mark.parametrize( + "configured,equivalent", + [ + ("gpt-5.6-luna", "system.ai.gpt-5-6-luna"), + ("system.ai.gpt-5-6-luna", "gpt-5.6-luna"), + ("gpt-6-luna", "system.ai.gpt-6-luna"), + ("system.ai.gpt-6-sol", "gpt-6-sol"), + ("glm-5-3", "system.ai.glm-5-3"), + ("system.ai.deepseek-v4-1-flash", "deepseek-v4-1-flash"), + ], +) +def test_codex_catalog_aliases_resolve_to_an_exact_available_model(config, configured, equivalent): + config.set_model(harness="codex", model=configured) + config.set_catalog(equivalent) + worker = config.cli("show", "--harness", "codex")["worker"] + assert worker == { + "model": equivalent, + "reasoning_effort": None, + "allow_inherited_fallback": False, + } + + +def test_codex_catalog_prefers_the_configured_spelling(config): + config.set_model(harness="codex", model="gpt-5.6-luna") + config.set_catalog("system.ai.gpt-5-6-luna", "gpt-5.6-luna") + assert config.cli("show", "--harness", "codex")["worker"]["model"] == "gpt-5.6-luna" + + +def test_codex_full_model_id_is_not_rewritten(config): + config.set_model(harness="codex", model="provider/custom-model") + assert config.cli("show", "--harness", "codex")["worker"] == { + "model": "provider/custom-model", + "reasoning_effort": None, + "allow_inherited_fallback": False, + } + + +@pytest.mark.parametrize( + "model", + [ + "provider/custom-model", + "provider:custom-model", + "system.ai.provider/custom-model", + "system.ai.provider:custom-model", + ], +) +def test_codex_custom_model_id_bypasses_catalog_alias_matching(config, model): + config.set_catalog("system.ai.gpt-5-6-luna", "provider/custom-model", "provider:custom-model") + config.set_model(harness="codex", model=model) + assert config.cli("show", "--harness", "codex")["worker"]["model"] == model + + +@pytest.mark.parametrize( + "configured,other_version", + [ + ("gpt-6-luna", "gpt-5.6-luna"), + ("gpt-5.6-luna", "gpt-6-luna"), + ("gpt-6-sol", "gpt-5.6-sol"), + ("gpt-5.6-sol", "gpt-6-sol"), + ], +) +def test_codex_catalog_aliases_do_not_cross_model_versions(config, configured, other_version): + config.set_model(harness="codex", model=configured) + config.set_catalog(other_version) + assert config.cli("show", "--harness", "codex")["worker"] == { + "model": configured, + "reasoning_effort": None, + "allow_inherited_fallback": False, + } + + +@pytest.mark.parametrize("available", [True, False]) +def test_codex_catalog_preserves_bundled_fallback_eligibility(config, available): + model = "system.ai.gpt-5-6-luna" if available else "another-model" + config.set_catalog(model) + roles = config.cli("show", "--harness", "codex") + assert set(roles) == {"explorer", "researcher", "worker", "tester", "reviewer"} + for choice in roles.values(): + assert choice == { + "model": model if available else "gpt-5.6-luna", + "reasoning_effort": "max", + "allow_inherited_fallback": True, + } + + +def test_codex_unavailable_role_does_not_block_an_available_role(config): + config.set_model(harness="codex", role="explorer", model="missing-model") + config.set_model(harness="codex", role="reviewer", model="glm-5-3") + config.set_catalog("system.ai.glm-5-3") + roles = config.cli("show", "--harness", "codex") + assert roles["explorer"] == { + "model": "missing-model", + "reasoning_effort": None, + "allow_inherited_fallback": False, + } + assert roles["reviewer"] == { + "model": "system.ai.glm-5-3", + "reasoning_effort": None, + "allow_inherited_fallback": False, + } + assert roles["worker"] == { + "model": "gpt-5.6-luna", + "reasoning_effort": "max", + "allow_inherited_fallback": True, + } + + +def test_codex_catalog_rejects_missing_or_malformed_catalog(config): + missing = config.set_catalog("system.ai.gpt-5-6-luna") + missing.unlink() + assert str(missing) in config.cli("show", "--harness", "codex", ok=False) + missing.write_text("not json") + assert str(missing) in config.cli("show", "--harness", "codex", ok=False) + + +@pytest.mark.parametrize( + "contents", ["[]", "{}", '{"models": []}', '{"models": [null]}', '{"models": [{"slug": 5}]}'] +) +def test_codex_catalog_rejects_invalid_structure(config, contents): + catalog = config.set_catalog("system.ai.gpt-5-6-luna") + catalog.write_text(contents) + assert str(catalog) in config.cli("show", "--harness", "codex", ok=False) + + +def test_codex_catalog_falls_back_to_codex_config(config): + catalog = config.set_catalog("system.ai.gpt-5-6-luna") + codex_home = config.root / "codex-home" + (codex_home / "ucode.config.toml").unlink() + (codex_home / "config.toml").write_text(f"model_catalog_json = {json.dumps(str(catalog))}\n") + config.env["CODEX_HOME"] = str(codex_home) + assert config.cli("show", "--harness", "codex")["worker"]["model"] == "system.ai.gpt-5-6-luna" + + +def test_codex_ug_catalog_takes_precedence_over_user_config(config): + config.set_catalog("system.ai.gpt-5-6-luna") + codex_home = Path(config.env["CODEX_HOME"]) + codex_home.mkdir(exist_ok=True) + (codex_home / "config.toml").write_text('model_catalog_json = "missing.json"\n') + assert config.cli("show", "--harness", "codex")["worker"]["model"] == "system.ai.gpt-5-6-luna" + + +def test_codex_catalog_config_relative_path(config): + codex_home = Path(config.env["CODEX_HOME"]) + codex_home.mkdir(exist_ok=True) + (codex_home / "models.json").write_text('{"models": [{"slug": "system.ai.gpt-5-6-luna"}]}') + (codex_home / "config.toml").write_text('model_catalog_json = "models.json"\n') + assert config.cli("show", "--harness", "codex")["worker"]["model"] == "system.ai.gpt-5-6-luna" + + +@pytest.mark.parametrize( + "contents", ["model_catalog_json = 5", 'model_catalog_json = ""', "not toml"] +) +def test_codex_catalog_rejects_invalid_config(config, contents): + codex_home = Path(config.env["CODEX_HOME"]) + codex_home.mkdir(exist_ok=True) + config_path = codex_home / "config.toml" + config_path.write_text(contents) + assert str(config_path) in config.cli("show", "--harness", "codex", ok=False) + + +def test_codex_catalog_does_not_hide_invalid_preferences(config): + config.set_catalog("system.ai.gpt-5-6-luna") + config.config.write_text('{"codex": {"worker": {"model": ""}}}') + assert "Invalid model for codex/worker" in config.cli("show", "--harness", "codex", ok=False) + + +@pytest.mark.parametrize("catalog", [None, "system.ai.gpt-5-6-luna", "another-model"]) +@pytest.mark.parametrize("user", [False, True]) +@pytest.mark.parametrize("model", ["gpt-5.6-luna", "provider/custom-model"]) +def test_codex_inherited_fallback_preserves_explicit_model_choices(config, user, model, catalog): + if catalog is not None: + config.set_catalog(catalog) + config.set_model(harness="codex", model=model, user=user) + roles = config.cli("show", "--harness", "codex") + expected = catalog if model == "gpt-5.6-luna" and catalog == "system.ai.gpt-5-6-luna" else model + assert roles["worker"]["model"] == expected + assert roles["worker"]["allow_inherited_fallback"] is False + assert roles["reviewer"]["allow_inherited_fallback"] is True + config.cli("unconfigure", user=user) + assert config.cli("show", "--harness", "codex")["worker"]["allow_inherited_fallback"] is True + + +def test_codex_fallback_stays_disabled_when_project_override_reveals_user_choice(config): + config.set_model(harness="codex", model="user-model", user=True) + config.set_model(harness="codex", model="project-model") + worker = config.cli("show", "--harness", "codex")["worker"] + assert worker["model"] == "project-model" + assert worker["allow_inherited_fallback"] is False + config.cli("unconfigure") + worker = config.cli("show", "--harness", "codex")["worker"] + assert worker["model"] == "user-model" + assert worker["allow_inherited_fallback"] is False + + +@pytest.mark.parametrize("role", ["explorer", "researcher", "worker", "tester", "reviewer"]) +def test_bundled_claude_agents_use_sonnet(config, role): + agent = config.script.parents[1] / "agents" / f"{role}.md" + frontmatter = agent.read_text().split("---", 2)[1] + assert "model: sonnet" in frontmatter.splitlines() + + +def test_bundled_skill_inherits_supervisor_model(config): + skill = config.script.parents[1] / "SKILL.md" + frontmatter = skill.read_text().split("---", 2)[1] + assert "model: inherit" in frontmatter.splitlines() + + +def test_full_id_effort_idempotence_and_cleanup(config): + arguments = ( + "set", + "--harness", + "claude", + "--role", + "worker", + "--model", + "provider/model:#id", + "--effort", + "high", + ) + config.cli(*arguments) + agent = config.agent() + assert 'model: "provider/model:#id"\neffort: high' in agent.read_text() + stamp = agent.stat().st_mtime_ns + assert config.cli(*arguments)["changed"] == [] + assert agent.stat().st_mtime_ns == stamp + assert config.cli("show", "--harness", "claude")["worker"] == { + "model": "provider/model:#id", + "effort": "high", + "subagent_type": "model-orchestrator-custom-project-worker", + } + config.cli("unconfigure") + assert not agent.exists() + assert not config.config.exists() + assert not config.journal.exists() + + +def test_project_overrides_user_and_harnesses_stay_separate(config): + config.set_model(model="user-model", user=True) + config.set_model(model="project-model") + config.set_model(harness="codex", role="reviewer", model="other-model") + assert config.cli("show", "--harness", "claude")["worker"]["model"] == "project-model" + original = config.agent(user=True).read_bytes() + config.agent(user=True).write_text("Edited but shadowed by project override") + assert config.cli("show", "--harness", "claude")["worker"]["model"] == "project-model" + config.agent(user=True).write_bytes(original) + assert config.cli("show", "--harness", "codex")["reviewer"] == { + "model": "other-model", + "reasoning_effort": None, + "allow_inherited_fallback": False, + } + config.cli("unconfigure") + assert config.cli("show", "--harness", "claude")["worker"]["model"] == "user-model" + config.cli("unconfigure", "--harness", "claude", user=True, ok=False) + assert config.agent(user=True).exists() + + +def test_edited_and_unowned_agents_are_preserved(config): + config.set_model() + agent = config.agent() + original = agent.read_text() + agent.write_text(original + "User edit\n") + before = config.config.read_bytes() + assert "Preserving" in config.cli("unconfigure", ok=False) + assert config.config.read_bytes() == before + assert agent.read_text().endswith("User edit\n") + assert "Missing/stale" in config.cli("show", "--harness", "claude", ok=False) + agent.write_text(original) + config.cli("unconfigure") + agent.write_text("Unrelated file\n") + assert "Preserving" in config.set_model(ok=False) + assert agent.read_text() == "Unrelated file\n" + assert not config.config.exists() + + +@pytest.mark.parametrize("model", ["", "bad\nmodel", "bad model", "bad\x01model"]) +def test_invalid_models_do_not_write(config, model): + config.set_model(model=model, ok=False) + assert not config.config.exists() + assert not config.agent().exists() + + +def test_invalid_effort_does_not_write(config): + config.cli( + "set", + "--harness", + "claude", + "--role", + "worker", + "--model", + "sonnet", + "--effort", + "unsupported-effort", + ok=False, + ) + assert not config.config.exists() + assert not config.agent().exists() + + +@pytest.mark.parametrize("location", ["project", "XDG_CONFIG_HOME", "CLAUDE_CONFIG_DIR"]) +def test_symlinked_scope_roots_are_supported(config, location): + alias = config.root / "alias" + if location == "project": + alias.symlink_to(config.project, target_is_directory=True) + config.project = alias + else: + target = Path(config.env[location]) + target.mkdir() + alias.symlink_to(target, target_is_directory=True) + config.env[location] = str(alias) + user = location != "project" + config.set_model(user=user) + assert ( + config.cli("show", "--harness", "claude", user=user)["worker"]["model"] + == "provider/custom-model" + ) + config.cli("unconfigure", user=user) + assert not config.agent(user=user).exists() + + +@pytest.mark.parametrize( + "location", + [ + ".claude", + ".model-orchestrator.json", + ".model-orchestrator.json.lock", + ".model-orchestrator.json.transaction.json", + ], +) +def test_managed_symlinks_are_rejected(config, location): + outside = config.root / "outside" + if location == ".claude": + outside.mkdir() + else: + outside.write_text("Untouched") + (config.project / location).symlink_to(outside) + assert "symlink" in config.set_model(ok=False) + if outside.is_dir(): + assert list(outside.iterdir()) == [] + else: + assert outside.read_text() == "Untouched" + + +@pytest.mark.parametrize( + "data,harness", + [ + ([], "claude"), + ({"unknown": {}}, "claude"), + ({"codex": {"unknown": {}}}, "codex"), + ({"_generated": {"worker": "bad"}}, "claude"), + ], +) +def test_malformed_configuration_is_rejected(config, data, harness): + config.config.write_text(json.dumps(data)) + config.cli("show", "--harness", harness, ok=False) + + +def test_forced_model_policy_is_preserved(config): + config.env.update(CLAUDE_CODE_SUBAGENT_MODEL="opus", CLAUDE_CODE_SUBAGENT_MODEL_FORCE="1") + assert "conflicts" in config.cli("show", "--harness", "claude", ok=False) + config.env.pop("CLAUDE_CODE_SUBAGENT_MODEL") + assert "parent model" in config.cli("show", "--harness", "claude", ok=False) + config.cli("show", "--harness", "codex") + + +def test_codex_update_preserves_edited_claude_agents(config): + config.set_model() + config.agent().write_text("User customization\n") + receipts = json.loads(config.config.read_text())["_generated"] + config.set_model(harness="codex", model="codex-model") + assert config.agent().read_text() == "User customization\n" + assert json.loads(config.config.read_text())["_generated"] == receipts + assert config.cli("show", "--harness", "codex")["worker"]["model"] == "codex-model" + assert "Preserving" in config.cli("unconfigure", ok=False) + + +@pytest.mark.parametrize( + "section,value", + [ + ("claude", {"worker": {"model": "invalid model"}}), + ("claude", []), + ("claude", {"unknown": None}), + ("_generated", {"worker": "bad"}), + ("_generated", []), + ("_generated", None), + ], +) +def test_codex_update_preserves_malformed_claude_state(config, section, value): + config.set_model() + agent_before = config.agent().read_bytes() + data = json.loads(config.config.read_text()) + data[section] = value + config.config.write_text(json.dumps(data)) + config.set_model(harness="codex", model="codex-model") + updated = json.loads(config.config.read_text()) + assert updated["claude"] == data["claude"] + assert updated["_generated"] == data["_generated"] + assert config.agent().read_bytes() == agent_before + assert config.cli("show", "--harness", "codex")["worker"]["model"] == "codex-model" + + +def test_codex_set_repairs_only_the_selected_role(config): + config.config.write_text(json.dumps({"codex": {"worker": {"model": 4}, "reviewer": []}})) + config.set_model(harness="codex", model="codex-model") + updated = json.loads(config.config.read_text()) + assert updated["codex"] == {"worker": {"model": "codex-model", "effort": None}, "reviewer": []} + + +@pytest.mark.parametrize( + "section,value", + [ + ("claude", {"worker": {"model": 4}}), + ("claude", []), + ("claude", {"unknown": {}}), + ("codex", {"worker": {"model": "invalid model"}}), + ("codex", []), + ], +) +def test_unconfigure_uses_ownership_receipts_despite_invalid_roles(config, section, value): + config.set_model() + unrelated = config.agent(role="reviewer") + unrelated.write_text("Unrelated file\n") + data = json.loads(config.config.read_text()) + data[section] = value + config.config.write_text(json.dumps(data)) + config.cli("unconfigure") + assert not config.config.exists() + assert not config.agent().exists() + assert unrelated.read_text() == "Unrelated file\n" + + +@pytest.mark.parametrize("receipts", [{"worker": "bad"}, [], None]) +def test_unconfigure_preserves_files_when_ownership_is_invalid(config, receipts): + config.set_model() + agent_before = config.agent().read_bytes() + data = json.loads(config.config.read_text()) + data["_generated"] = receipts + config.config.write_text(json.dumps(data)) + config_before = config.config.read_bytes() + assert "ownership" in config.cli("unconfigure", ok=False) + assert config.config.read_bytes() == config_before + assert config.agent().read_bytes() == agent_before + + +@pytest.mark.parametrize("action", ["set", "unconfigure"]) +def test_corrupt_json_updates_and_cleanup_preserve_files(config, action): + config.set_model() + agent_before = config.agent().read_bytes() + config.config.write_text("{invalid json") + if action == "set": + error = config.set_model(harness="codex", ok=False) + else: + error = config.cli("unconfigure", ok=False) + assert "repair or restore" in error + assert config.config.read_text() == "{invalid json" + assert config.agent().read_bytes() == agent_before + + +@pytest.mark.parametrize("user", [False, True]) +@pytest.mark.parametrize("edited", [False, True]) +@pytest.mark.parametrize("role", ["explorer", "researcher", "worker", "tester", "reviewer"]) +def test_template_upgrade_refreshes_only_unedited_owned_agents(config, user, edited, role): + package = config.root / "plugin" + shutil.copytree(config.script.parents[1], package) + config.script = package / "scripts/configure.py" + arguments = ( + "set", + "--harness", + "claude", + "--role", + role, + "--model", + "provider/custom-model", + "--effort", + "high", + ) + config.cli(*arguments, user=user) + config_path = config.user_config if user else config.config + config_before = config_path.read_bytes() + template = package / "agents" / f"{role}.md" + template.chmod(template.stat().st_mode | stat.S_IWUSR) + template.write_text(template.read_text() + "\nUpdated role instructions.\n") + agent = config.agent(role=role, user=user) + if edited: + agent.write_text(agent.read_text() + "User edit\n") + assert "Missing/stale" in config.cli("show", "--harness", "claude", user=user, ok=False) + if edited: + assert "Preserving" in config.cli(*arguments, user=user, ok=False) + assert config_path.read_bytes() == config_before + assert agent.read_text().endswith("User edit\n") + else: + config.cli(*arguments, user=user) + assert agent.read_text().endswith("Updated role instructions.\n") + assert ( + json.loads(config_path.read_text())["_generated"] + != json.loads(config_before)["_generated"] + ) + resolved = config.cli("show", "--harness", "claude", user=user)[role] + assert resolved["model"] == "provider/custom-model" + assert resolved["effort"] == "high" + + +@pytest.mark.parametrize( + "choice", + [{"model": "invalid model"}, {"model": 4}, {"model": "sonnet", "effort": "invalid"}, []], +) +def test_invalid_shadowed_role_is_ignored_but_invalid_fallback_is_rejected(config, choice): + config.set_model(model="user-model", user=True) + config.set_model(model="project-model") + data = json.loads(config.user_config.read_text()) + data["claude"]["worker"] = choice + config.user_config.write_text(json.dumps(data)) + assert config.cli("show", "--harness", "claude")["worker"]["model"] == "project-model" + config.cli("show", "--harness", "claude", user=True, ok=False) + config.cli("show", "--harness", "codex") + + +def test_corrupt_user_json_is_not_hidden_by_project_overrides(config): + config.set_model(user=True) + config.set_model(model="project-model") + config.user_config.write_text("{invalid json") + config.cli("show", "--harness", "claude", ok=False) + + +@pytest.mark.parametrize("action", ["show", "set", "unconfigure"]) +def test_preference_operations_wait_for_existing_writer(config, action): + config.cli( + "set", "--harness", "claude", "--role", "worker", "--model", "sonnet", "--effort", "xhigh" + ) + before = config.config.read_bytes() + agent_before = config.agent().read_bytes() + lock = config.config.with_name(config.config.name + ".lock") + ready = config.root / "preference-operation-ready" + arguments = [action] if action == "unconfigure" else [action, "--harness", "claude"] + if action == "set": + arguments += ["--role", "reviewer", "--model", "sonnet"] + # Finish routing setup before timing the wait for the configuration lock. + program = """ +import runpy +import sys +from pathlib import Path +script, ready, *arguments = sys.argv[1:] +module = runpy.run_path(script) +module["require_enabled"]() +sys.argv = [script, *arguments] +Path(ready).touch() +module["main"]() +""" + process = None + try: + with lock.open("a+b") as stream: + acquire_exclusive_file_lock(stream) + try: + process = subprocess.Popen( + [ + sys.executable, + "-c", + program, + str(config.script), + str(ready), + *arguments, + "--project", + str(config.project), + ], + env=config.env, + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + ) + deadline = time.monotonic() + 20 + while not ready.exists(): + assert process.poll() is None, process.communicate(timeout=20) + assert time.monotonic() < deadline, "Preference helper did not start" + time.sleep(0.01) + with pytest.raises(subprocess.TimeoutExpired): + process.communicate(timeout=0.5) + assert config.config.read_bytes() == before + assert config.agent().read_bytes() == agent_before + assert not config.agent(role="reviewer").exists() + finally: + release_file_lock(stream) + stdout, stderr = process.communicate(timeout=20) + assert process.returncode == 0, stderr + result = json.loads(stdout) + if action == "show": + assert result["worker"]["model"] == "sonnet" + assert config.config.read_bytes() == before + elif action == "set": + preferences = json.loads(config.config.read_text())["claude"] + assert set(preferences) == {"worker", "reviewer"} + assert preferences["worker"]["effort"] == "xhigh" + assert config.agent(role="reviewer").is_file() + assert config.agent().read_bytes() == agent_before + else: + assert not config.config.exists() + assert not config.agent().exists() + finally: + if process is not None and process.poll() is None: + process.kill() + process.communicate(timeout=20) + + +def test_show_reads_existing_lock_on_read_only_mount(config, configure_module, monkeypatch, capsys): + config.cli( + "set", + "--harness", + "codex", + "--role", + "worker", + "--model", + "gpt-6-luna", + "--effort", + "max", + user=True, + ) + lock = config.user_config.with_name(config.user_config.name + ".lock") + assert lock.is_file() + for name in ("XDG_CONFIG_HOME", "CLAUDE_CONFIG_DIR", "CODEX_HOME"): + monkeypatch.setenv(name, config.env[name]) + original_open = Path.open + + # Reject writes to the lock file to represent a read-only configuration mount. + def open_without_lock_writes(path, mode="r", *args, **kwargs): + if path == lock and "a" in mode: + raise OSError(errno.EROFS, "Read-only file system", str(path)) + return original_open(path, mode, *args, **kwargs) + + monkeypatch.setattr(Path, "open", open_without_lock_writes) + monkeypatch.setattr( + sys, + "argv", + [str(config.script), "show", "--harness", "codex", "--project", str(config.project)], + ) + configure_module.main() + assert json.loads(capsys.readouterr().out)["worker"] == { + "model": "gpt-6-luna", + "reasoning_effort": "max", + "allow_inherited_fallback": False, + } + + +def test_legacy_lock_directory_has_actionable_error(config): + lock = config.config.with_name(config.config.name + ".lock") + lock.mkdir() + assert "Legacy lock directory" in config.set_model(ok=False) + assert not config.config.exists() + + +def test_failed_config_write_restores_prior_agent(config, configure_module, monkeypatch): + config.set_model(model="old-model") + before_agent, before_config = config.agent().read_bytes(), config.config.read_bytes() + previous = configure_module.load_config(config.config) + desired = {"claude": {"worker": {"model": "new-model"}}} + original_replace = os.replace + + def fail_config(source, destination): + if Path(destination) == config.config: + raise OSError("simulated full disk") + original_replace(source, destination) + + monkeypatch.setattr(os, "replace", fail_config) + with pytest.raises(OSError, match="full disk"): + configure_module.update_scope(config.project, previous, desired) + assert config.agent().read_bytes() == before_agent + assert config.config.read_bytes() == before_config + assert not config.journal.exists() + + +@pytest.mark.parametrize("action", ["create", "update", "unconfigure"]) +@pytest.mark.parametrize("boundary", ["agent", "config"]) +def test_interrupted_writes_are_recovered_and_locks_released(config, action, boundary): + if action != "create": + config.set_model(model="old-model") + before_config = config.config.read_bytes() if config.config.exists() else None + before_agent = config.agent().read_bytes() if config.agent().exists() else None + arguments = ( + ("unconfigure",) + if action == "unconfigure" + else ( + "set", + "--harness", + "claude", + "--role", + "worker", + "--model", + "new-model", + ) + ) + config.interrupt(config.agent() if boundary == "agent" else config.config, *arguments) + assert config.journal.exists() + resolved = config.cli("show", "--harness", "claude") + assert resolved["worker"]["model"] == ("sonnet" if action == "create" else "old-model") + assert (config.config.read_bytes() if config.config.exists() else None) == before_config + assert (config.agent().read_bytes() if config.agent().exists() else None) == before_agent + assert not config.journal.exists() + config.set_model(model="next-model") + + +def test_interrupted_recovery_can_be_retried(config): + config.set_model(model="old-model") + config.interrupt( + config.agent(), "set", "--harness", "claude", "--role", "worker", "--model", "new-model" + ) + config.interrupt(config.agent(), "show", "--harness", "claude") + assert config.journal.exists() + assert config.cli("show", "--harness", "claude")["worker"]["model"] == "old-model" + assert not config.journal.exists() + + +@pytest.mark.parametrize("edited_file", ["agent", "config"]) +def test_recovery_preserves_post_crash_edits(config, edited_file): + config.set_model(model="old-model") + config.interrupt( + config.agent(), "set", "--harness", "claude", "--role", "worker", "--model", "new-model" + ) + target = config.agent() if edited_file == "agent" else config.config + target.write_text("Post-crash user edit") + before_agent = config.agent().read_bytes() + before_config = config.config.read_bytes() + assert "Preserving edited file during recovery" in config.cli( + "show", "--harness", "claude", ok=False + ) + assert config.agent().read_bytes() == before_agent + assert config.config.read_bytes() == before_config + assert config.journal.exists() + + +def test_recovery_rejects_unmanaged_paths(config): + outside = config.root / "outside" + outside.write_text("Untouched") + config.journal.write_text( + json.dumps( + { + "config": {"before": None, "after": None}, + "../outside": {"before": None, "after": outside.read_bytes().hex()}, + } + ) + ) + assert "Invalid recovery journal" in config.cli("show", "--harness", "claude", ok=False) + assert outside.read_text() == "Untouched" diff --git a/tests/test_orchestrator_legacy_plugins.py b/tests/test_orchestrator_legacy_plugins.py new file mode 100644 index 000000000..a5761025d --- /dev/null +++ b/tests/test_orchestrator_legacy_plugins.py @@ -0,0 +1,194 @@ +"""The marketplace predecessor must not bypass UG's shared routing controls.""" + +import json +import tomllib + +import pytest +import tomlkit + +from ucode import codex_config +from ucode.agents import claude +from ucode.smart_routing import orchestrator, v2 + + +@pytest.mark.parametrize("launch_mode", ["direct", "relayed", "routing"]) +@pytest.mark.parametrize("routing_enabled", ["0", "1"]) +def test_claude_suppresses_legacy_plugins_without_changing_saved_settings( + tmp_path, monkeypatch, launch_mode, routing_enabled +): + monkeypatch.setenv(v2.ENABLE_SMART_ROUTING_ENV_VAR, routing_enabled) + user_plugin = "model-orchestrator@eng-plugin-marketplace-experimental" + project_plugin = "model-orchestrator@project-marketplace" + caller_plugin = "model-orchestrator@caller-marketplace" + user_settings = claude.CLAUDE_USER_SETTINGS_PATH + registry = user_settings.parent / "plugins" / "installed_plugins.json" + registry.parent.mkdir(parents=True) + registry.write_text( + json.dumps({"version": 2, "plugins": {project_plugin: [{"scope": "project"}]}}) + ) + user_settings.write_text( + json.dumps({"enabledPlugins": {user_plugin: True, "unrelated@marketplace": True}}) + ) + gateway_settings = tmp_path / "ucode-settings.json" + gateway_settings.write_text(json.dumps({"apiKeyHelper": "gateway-helper"})) + monkeypatch.setattr(claude, "CLAUDE_SETTINGS_PATH", gateway_settings) + user_hook = {"hooks": [{"type": "command", "command": "user-policy"}]} + caller = tmp_path / "caller-settings.json" + caller.write_text( + json.dumps( + { + "enabledPlugins": { + caller_plugin: True, + "unrelated@marketplace": True, + "model-orchestrator-extra@marketplace": True, + }, + "hooks": {"UserPromptSubmit": [user_hook]}, + } + ) + ) + before = { + path: path.read_bytes() for path in (user_settings, registry, gateway_settings, caller) + } + args = ["--settings", str(caller), "--print", "hello"] + + if launch_mode == "routing": + composed, remaining = claude._compose_v2_settings(args) + assert composed["enabledPlugins"][user_plugin] is False + argv = claude._build_claude_argv("claude", remaining, settings_override=composed) + else: + argv = claude._build_claude_argv("claude", args, relayed=launch_mode == "relayed") + + settings = json.loads(argv[argv.index("--settings") + 1]) + assert settings["enabledPlugins"] == { + user_plugin: False, + project_plugin: False, + caller_plugin: False, + "unrelated@marketplace": True, + "model-orchestrator-extra@marketplace": True, + } + assert settings["apiKeyHelper"] == "gateway-helper" + assert settings["hooks"]["UserPromptSubmit"] == [user_hook] + assert argv[-2:] == ["--print", "hello"] + assert args == ["--settings", str(caller), "--print", "hello"] + assert {path: path.read_bytes() for path in before} == before + + +@pytest.mark.parametrize("custom_home", [False, True]) +def test_claude_suppresses_installed_plugin_without_caller_settings( + tmp_path, monkeypatch, custom_home +): + config_dir = ( + tmp_path / "custom-claude" if custom_home else claude.CLAUDE_USER_SETTINGS_PATH.parent + ) + registry = config_dir / "plugins" / "installed_plugins.json" + registry.parent.mkdir(parents=True) + name = "model-orchestrator@custom-marketplace" + registry.write_text(json.dumps({"plugins": {name: [{"scope": "user"}]}})) + if custom_home: + monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(config_dir)) + gateway_settings = tmp_path / "ucode-settings.json" + gateway_settings.write_text("{}") + monkeypatch.setattr(claude, "CLAUDE_SETTINGS_PATH", gateway_settings) + + argv = claude._build_claude_argv("claude", ["--print", "hello"]) + + settings = json.loads(argv[argv.index("--settings") + 1]) + assert settings == {"enabledPlugins": {name: False}} + assert gateway_settings.read_text() == "{}" + + +def test_claude_without_legacy_plugin_keeps_settings_file_argument(tmp_path, monkeypatch): + user_settings = claude.CLAUDE_USER_SETTINGS_PATH + user_settings.parent.mkdir(parents=True) + user_settings.write_text( + json.dumps({"enabledPlugins": {"model-orchestrator-extra@marketplace": True}}) + ) + gateway_settings = tmp_path / "ucode-settings.json" + gateway_settings.write_text("{}") + monkeypatch.setattr(claude, "CLAUDE_SETTINGS_PATH", gateway_settings) + + assert claude._build_claude_argv("claude", ["--print", "hello"]) == [ + "claude", + "--settings", + str(gateway_settings), + "--print", + "hello", + ] + + +@pytest.mark.parametrize("custom_home", [False, True]) +def test_codex_suppresses_legacy_plugins_from_saved_config_layers( + tmp_path, monkeypatch, custom_home +): + default_profile = codex_config.DEFAULT_CODEX_CONFIG_PATH + profile = default_profile + if custom_home: + monkeypatch.setenv("CODEX_HOME", str(tmp_path / "custom-codex")) + profile = tmp_path / "custom-codex" / "ucode.config.toml" + default_profile.parent.mkdir(parents=True) + default_profile.write_text('[plugins."model-orchestrator@unused-home"]\nenabled = true\n') + managed = tmp_path / "managed_config.toml" + monkeypatch.setattr(codex_config, "codex_managed_config_path", lambda: managed) + names = { + managed: "model-orchestrator@managed-marketplace", + profile: "model-orchestrator@profile-marketplace", + profile.parent / "config.toml": "model-orchestrator@isaac-sync-user.marketplace", + } + for path, name in names.items(): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text( + tomlkit.dumps( + { + "plugins": { + name: {"enabled": True}, + "unrelated@marketplace": {"enabled": True}, + "model-orchestrator-extra@marketplace": {"enabled": True}, + } + } + ) + ) + before = {path: path.read_bytes() for path in names} + + override = orchestrator.legacy_codex_plugin_config(default_profile) + + assert override == {"plugins": {name: {"enabled": False} for name in names.values()}} + flag, value = codex_config.codex_config_args(override) + assert flag == "--config" + assert value.startswith("plugins={") + assert tomllib.loads(value) == override + assert {path: path.read_bytes() for path in before} == before + + +def test_codex_suppresses_project_plugins_from_nested_directory(tmp_path, monkeypatch): + monkeypatch.setenv("CODEX_HOME", str(tmp_path / "user")) + project = tmp_path / "project" + nested = project / "nested" + names = { + project / ".codex/config.toml": "model-orchestrator@project", + nested / ".codex/config.toml": "model-orchestrator@nested", + } + for path, name in names.items(): + path.parent.mkdir(parents=True) + path.write_text( + tomlkit.dumps( + { + "plugins": {name: {"enabled": True}, "unrelated@project": {"enabled": True}}, + "hooks": {"UserPromptSubmit": [{"hooks": [{"command": "project-policy"}]}]}, + } + ) + ) + before = {path: path.read_bytes() for path in names} + + override = orchestrator.legacy_codex_plugin_config(cwd=nested) + + assert override == {"plugins": {name: {"enabled": False} for name in names.values()}} + assert {path: path.read_bytes() for path in before} == before + + +def test_codex_without_legacy_plugin_has_no_override(tmp_path, monkeypatch): + monkeypatch.setenv("CODEX_HOME", str(tmp_path)) + (tmp_path / "config.toml").write_text( + '[plugins."model-orchestrator-extra@marketplace"]\nenabled = true\n' + ) + + assert orchestrator.legacy_codex_plugin_config() == {} diff --git a/tests/test_smart_router.py b/tests/test_smart_router.py index 81613c882..e9e687d2d 100644 --- a/tests/test_smart_router.py +++ b/tests/test_smart_router.py @@ -12,7 +12,7 @@ from typer.testing import CliRunner from ucode import cli, config_io, skills -from ucode.skills import SMART_ROUTER_SKILL +from ucode.skills import ORCHESTRATOR_SKILL, SMART_ROUTER_SKILL from ucode.smart_routing import session_env, v2 runner = CliRunner() @@ -33,6 +33,8 @@ def test_smart_routed_session_installs_skill(tmp_path, monkeypatch, agent): home = config_io.APP_DIR.parent assert home.joinpath(f".{agent}/skills/{SMART_ROUTER_SKILL}/SKILL.md").is_file() + assert home.joinpath(f".{agent}/skills/{ORCHESTRATOR_SKILL}/SKILL.md").is_file() + assert home.joinpath(f".{agent}/skills/{ORCHESTRATOR_SKILL}/scripts/configure.py").is_file() assert not home.joinpath(f".agents/skills/{SMART_ROUTER_SKILL}").exists() assert session_path == Path(os.environ[session_env.SESSION_ENV_VAR]) assert Path(os.environ[session_env.SESSION_ENV_VAR]).is_file() From 2e085825636247021b620d8b1a6e63384cc98f30 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 16:44:54 +0000 Subject: [PATCH 02/25] Let smart routing choose orchestrator models --- README.md | 10 +- justfile | 8 +- skills/orchestrate/README.md | 40 +- skills/orchestrate/SKILL.md | 140 ++-- skills/orchestrate/agents/explorer.md | 2 +- skills/orchestrate/agents/researcher.md | 2 +- skills/orchestrate/agents/reviewer.md | 2 +- skills/orchestrate/agents/tester.md | 2 +- skills/orchestrate/agents/worker.md | 2 +- skills/orchestrate/scripts/configure.py | 438 ----------- src/ucode/smart_routing/orchestrator.py | 18 +- tests/README.md | 12 +- tests/integration/README.md | 8 +- tests/test_orchestrator.py | 72 +- tests/test_orchestrator_config.py | 954 ------------------------ tests/test_smart_router.py | 25 +- 16 files changed, 201 insertions(+), 1534 deletions(-) delete mode 100644 skills/orchestrate/scripts/configure.py delete mode 100644 tests/test_orchestrator_config.py diff --git a/README.md b/README.md index 453e5fae3..14f61bfbb 100644 --- a/README.md +++ b/README.md @@ -256,7 +256,7 @@ the root. Orchestration follows the existing smart-routing launch eligibility and session controls; it has no separate rollout flag. Turning Smart Router off stops new -automatic delegation, including fallback to default role models. Turning it on +automatic delegation. Turning it on restores orchestration. Explicit user requests for subagents still use normal harness behavior while routing is off. Stored skill files do not activate it in later non-routed sessions. The existing Isaac pilot gate and UG launch exclusions @@ -267,10 +267,10 @@ including Isaac-synced Codex registrations, for each Claude or Codex launch. This also applies when smart routing is off, so the old hooks cannot activate orchestration independently. Codex project registrations are included. Routed Codex launches use its native configuration resolver to preserve trusted project hooks; -untrusted project hooks stay disabled. Saved plugin settings, unrelated plugins and hooks, -and existing role preferences are retained. See the bundled -[orchestrator documentation](skills/orchestrate/README.md) for configuration and -cutover details. +untrusted project hooks stay disabled. Saved plugin settings and unrelated plugins +and hooks remain intact. Smart routing selects subagent models; separate role-model +preferences are ignored and their files are left untouched. See the bundled +[orchestrator documentation](skills/orchestrate/README.md) for cutover details. ## Managed Files diff --git a/justfile b/justfile index a3511b450..f1fb8558f 100644 --- a/justfile +++ b/justfile @@ -12,11 +12,11 @@ lint: ruff-check ruff-format-check ty # Lint without fixing (CI-equivalent). ruff-check: - uv run ruff check src/ tests/ skills/ + uv run ruff check src/ tests/ # Verify formatting without writing files (CI-equivalent). ruff-format-check: - uv run ruff format --check src/ tests/ skills/ + uv run ruff format --check src/ tests/ # Type-check the package. ty: @@ -24,5 +24,5 @@ ty: # Autofix lint + format in place. Local convenience; not part of the gate. fix: - uv run ruff check --fix src/ tests/ skills/ - uv run ruff format src/ tests/ skills/ + uv run ruff check --fix src/ tests/ + uv run ruff format src/ tests/ diff --git a/skills/orchestrate/README.md b/skills/orchestrate/README.md index c8d4da4c8..afdf70939 100644 --- a/skills/orchestrate/README.md +++ b/skills/orchestrate/README.md @@ -1,20 +1,21 @@ # UG model orchestrator -UG bundles the `orchestrate` workflow, five Claude role definitions, and the -existing role-preference helper from `model-orchestrator` 0.4.10. Smart-routed +UG bundles the `orchestrate` workflow and five Claude role definitions from +`model-orchestrator` 0.4.10. Smart-routed Claude and Codex launches install this skill alongside `smart-router`. The workflow is injected before root prompts and after compaction. Its activation -and model-resolution checks require a UG smart-routing session and read the same +and pre-delegation checks require a UG smart-routing session and read the same session controls as the routing hooks. Turning Smart Router off stops new automatic delegation and supersedes the previous workflow. Turning it on restores both features. An installed skill or saved model preference cannot enable them. Explicit user requests for subagents still use native harness behavior while routing -is off, without the orchestrator's model-resolution helper or role models. +is off, without the orchestrator's workflow or routing check. User instructions take precedence, and easy tasks remain in the root. Claude loads the bundled roles as `ug-smart-router:` in its temporary -routing plugin. Codex uses native spawning with per-call model preferences. +routing plugin. Both agents delegate without model or reasoning-effort overrides; +the routing hook selects the model. Role instructions belong in each task prompt because routing may replace the requested Claude role or Codex model. Hook approval in the native `/hooks` UI is still required where the harness prompts for it. @@ -30,21 +31,16 @@ unrelated plugins and hooks, and launches outside UG are unaffected. This covers marketplace installations; manually copied activation hooks or development copies passed through `--plugin-dir` need to be removed separately. -Existing `.model-orchestrator.json` project preferences and -`$XDG_CONFIG_HOME/model-orchestrator/config.json` user preferences keep their -format and precedence. Claude custom agent names and ownership hashes are -unchanged. Bundled defaults remain Sonnet for Claude and `gpt-5.6-luna` at `max` -effort for Codex; routing determines the final model. The helper reads Codex's -catalog using UG's managed, profile, then user config precedence, including -`CODEX_HOME`, rather than Isaac's catalog environment variable. - -The [skill](SKILL.md) documents `show`, `set`, and `unconfigure`. Run its helper -with the launching `UCODE_SMART_ROUTER_PYTHON`, not an arbitrary Python on PATH. -Only `show` requires enabled routing; changing or removing preferences does not -activate orchestration. Configuration retains the original ownership checks -and interrupted-write recovery. Preference operations use UG's existing file -lock and wait for an operation in the same scope to finish. User-edited agents -are preserved and reported for reconciliation. +Separate role-model preferences are not used by the UG workflow. Existing +`.model-orchestrator.json` project preferences, +`$XDG_CONFIG_HOME/model-orchestrator/config.json` user preferences, and generated +custom Claude agents are left untouched. The bundled workflow uses the router's +model selection and requires no preference setup, locking, or recovery. + +The [skill](SKILL.md) checks eligibility with +`"$UCODE_SMART_ROUTER_PYTHON" -m ucode.smart_routing.orchestrator --check` before +delegating. This command reads session controls without reading or writing +preference files, succeeds silently when enabled, and exits nonzero when disabled. ## Attribution @@ -52,5 +48,5 @@ Migrated from the Databricks `model-orchestrator` plugin 0.4.10 by Arnav Singhvi Originally adapted from [donvito/codex-astra-luna-orchestrator](https://github.com/donvito/codex-astra-luna-orchestrator/tree/21710352ec201f8634874d8298e0eca694e298a8) under Apache-2.0; see [LICENSE.upstream](LICENSE.upstream). UG changes add shared -routing-state checks, launch-scoped activation and Claude roles, UG catalog -discovery, and cross-platform locking. +routing-state checks and launch-scoped activation and Claude roles, and delegate +model selection to smart routing. diff --git a/skills/orchestrate/SKILL.md b/skills/orchestrate/SKILL.md index 0509b3a37..fd0a51de8 100644 --- a/skills/orchestrate/SKILL.md +++ b/skills/orchestrate/SKILL.md @@ -2,9 +2,9 @@ name: orchestrate description: Coordinate substantive development with native subagents while Unity Gateway smart routing is enabled. Follow the routing-state check before using this workflow. Skip easy tasks and explicit no-subagent requests. model: inherit -argument-hint: "[task, configure, or unconfigure]" +argument-hint: "[task]" metadata: - version: "1.0.0" + version: "1.1.0" --- # Model orchestrator @@ -12,19 +12,26 @@ metadata: ## Smart-routing gate This workflow is active only in a UG-launched smart-routing session while routing -is enabled. Installed skill files, old context, and model preferences do not -enable it. Before **every new delegation under this workflow, including a retry**, -run the resolution command below with the launching `$UCODE_SMART_ROUTER_PYTHON` -interpreter. It checks the same session controls as the routing hooks. If the -interpreter or session marker is absent, or resolution reports routing off, do -not use this workflow. -Do not set routing flags or create a session to bypass this check. +is enabled. Installed skill files and old context do not enable it. Before +**every new delegation under this workflow**, check the same session controls +as the routing hooks with the launching interpreter: + +```text +"$UCODE_SMART_ROUTER_PYTHON" -m ucode.smart_routing.orchestrator --check +``` + +The command succeeds silently when routing is enabled. In PowerShell, use +`& $env:UCODE_SMART_ROUTER_PYTHON` in place of `"$UCODE_SMART_ROUTER_PYTHON"`. +If the interpreter is absent or the command fails, do not use this workflow; +report the problem and continue authorized work in the root. Never choose +another Python from PATH, set routing flags, or create a session to bypass +this check. Turning Smart Router off also turns this workflow off immediately and supersedes earlier orchestration instructions. Do not start new automatic delegation or use orchestrator role models as a fallback. Continue in the root unless the user explicitly requests a subagent; honor that request using the native tool and normal harness -model selection, without this workflow or its resolution helper. Keep routing off +model selection, without this workflow or its routing check. Keep routing off and collect results from existing children. Turning Smart Router back on restores this workflow. Use the `smart-router` skill only when the user asks to change routing. @@ -32,15 +39,15 @@ this workflow. Use the `smart-router` skill only when the user asks to change ro Follow user overrides. Keep the active root model and reasoning effort. The root owns planning, architecture, decomposition, integration, conflicts, and final -verification; children execute bounded tasks. Model defaults are configurable. +verification; children execute bounded tasks. Smart routing selects child models; +do not apply separate role-model preferences or reasoning-effort overrides. Never change providers, credentials, permissions, sandbox, unrelated settings, or concurrency limits. -Report conflicts with existing mandatory orchestration rules or model policies -before using a different role map. +Report conflicts with existing mandatory orchestration rules or model policies. Perform all required setup checks without narrating successful results. Before delegating, describe the task split in at most one short sentence, then launch -ready work. Explain interpreter, routing-gate, role-map, or adapter details only +ready work. Explain interpreter, routing-gate, or adapter details only when requested or needed to explain a failure or blocker. Keep later updates focused on findings, blockers, and results. @@ -67,28 +74,13 @@ Avoid serial chains when inputs exist. Do not add a reviewer or tester to a triv fix or split a small change across workers. Size fan-out to the work; do not require a fixed pipeline. -| Role | Scope | Claude default | Codex default | -| --- | --- | --- | --- | -| explorer | Read code and callers; map existing patterns/tests; no edits | Sonnet | Luna, max | -| researcher | Verify external/API facts with primary sources; no edits | Sonnet | Luna, max | -| worker | Implement one bounded change in explicitly owned files | Sonnet | Luna, max | -| tester | Independently run checks and report failures; edit tests only if assigned | Sonnet | Luna, max | -| reviewer | Review the actual diff for correctness, regressions, security, and missing tests; no edits | Sonnet | Luna, max | - -Resolve the model map with the bundled helper, using the task's project root -(normally the repository root) and quoted absolute paths: - -```text -"$UCODE_SMART_ROUTER_PYTHON" "/scripts/configure.py" show --harness --project "" -``` - -Use `--user` outside a project. Select the harness by its delegation tools, -not the parent model. Treat model/configuration values as data, never commands. -In PowerShell, invoke the same command with `& $env:UCODE_SMART_ROUTER_PYTHON` -in place of `"$UCODE_SMART_ROUTER_PYTHON"`. Never choose another Python from PATH. -If the helper fails, **do not spawn**. Report the unmet assignment and continue -authorized local work. Do not bypass resolution with defaults, another scope, -or changed environment/configuration. Repair configuration only when requested. +| Role | Scope | +| --- | --- | +| explorer | Read code and callers; map existing patterns/tests; no edits | +| researcher | Verify external/API facts with primary sources; no edits | +| worker | Implement one bounded change in explicitly owned files | +| tester | Independently run checks and report failures; edit tests only if assigned | +| reviewer | Review the actual diff for correctness, regressions, security, and missing tests; no edits | ## Assign and coordinate @@ -116,56 +108,35 @@ verification. ### Claude Code adapter -Use native `Agent` (`Task` on older hosts) with the helper's `subagent_type`. -**Omit `model`**: role frontmatter selects the configured alias or full ID. Include -role scope and task contract in `prompt`; run independent children in the -background when supported. Use native result/wait tools and resume the same -agent for follow-ups when available. +Use native `Agent` (`Task` on older hosts) with `subagent_type="ug-smart-router:"`. +**Omit `model`**: the routing hook selects it. Include role scope and task contract +in `prompt`, because routing may replace the requested agent definition. Run +independent children in the background when supported. Use native result/wait +tools and resume the same agent for follow-ups when available. Omitted role tool lists inherit parent tools, including deferred MCP tools; parent permissions and hooks still apply. Read-only scope is instructional. -Configured agents have distinct names. Report missing definitions as requiring -reload/restart; do not substitute built-ins. Per-call model overrides are alias-only -on the tested host; custom IDs belong in definitions. Managed forced-model policy -takes precedence; report conflicts without clearing it. +Report missing bundled definitions as requiring reload/restart; do not substitute +custom agents with saved model preferences. Managed forced-model policy takes +precedence; report conflicts without clearing it. ### Codex adapter -Make the initial native `spawn_agent` call with the helper's `model`. Pass -`reasoning_effort` only when non-null; otherwise omit it to use the native default. -The helper resolves equivalent spellings against the active catalog when available. -Attempt its model even if absent from the tool's partial preview. Do not retry -another spelling, invent aliases, or substitute a successor. Send the role scope -and contract in `message`; use `fork_turns="none"` if overrides require fresh context. Use the +Use native `spawn_agent` without `model` or `reasoning_effort` overrides. The routing +hook selects the model. Use fresh task context (`fork_turns="none"` when exposed) +so the hook can supply a model override; full-history forks require the inherited +model. Send the needed context, role scope, and contract in `message`; use the host's native follow-up, message, wait, and close tools. Do not choose a custom -role that pins a different model or effort. +role that pins a model or effort. Native spawning needs no role TOMLs or global `[agents]` defaults. Never simulate delegation with nested CLIs. Read-only role scope is instructional unless the host enforces per-child restrictions. -### Recover a Codex delegation - -Before every retry, check these conditions in order: - -1. Did `spawn_agent` return a child ID for this assignment? If yes, **never spawn - a replacement**, even after closing it. An error from wait, notification, or - the child provider is a child failure, not a rejected spawn. Report it unmet. -2. Is the error permission, authentication, or capacity related? Stop. No alias - retry, inherited fallback, or changes to permissions, credentials, or limits. -3. Did `spawn_agent` itself reject the model/effort before returning any child ID? - Only this selection failure (or a schema without overrides) permits recovery. - -Require `allow_inherited_fallback: true` from successful resolution for the -assigned role (bundled defaults only). Honor explicit settings and conversation/ -policy constraints; never change roles or configuration to evade them. - -If eligible and the routing-state check still passes, disclose the failure and -**attempt one native spawn omitting both -`model` and `reasoning_effort`**, with the same contract and fresh context (`fork_turns="none"` -when exposed). The routing hook selects the model. Do not assume routing ran or -fallback will succeed. Never use this retry when routing is off. If forbidden -or unsuccessful, stop retrying and report the error and unmet assignment. +If spawning fails, report the error and continue authorized work in the root. +Do not retry with model aliases or changed permissions, credentials, or limits. +Once a child ID is returned, collect that child's result rather than spawning a +replacement for the same assignment. ## Integrate and verify @@ -180,22 +151,5 @@ respawning. Report unavailable models, tools, and substitutions. Finish with the concrete result, verification actually performed, and material remaining limitations. Do not claim cost or speed improvements without measurements. -## Configure / unconfigure - -Only change preferences when requested. The helper supports: - -```text -"$UCODE_SMART_ROUTER_PYTHON" "/scripts/configure.py" set --harness --role --model [--effort ] <--project |--user> -"$UCODE_SMART_ROUTER_PYTHON" "/scripts/configure.py" unconfigure <--project |--user> -``` - -User defaults live in `$XDG_CONFIG_HOME/model-orchestrator/config.json` (normally -`~/.config`); project-root `.model-orchestrator.json` overrides them. `set` without -`--effort` uses native defaults; Claude inherits session effort if unset. -Refresh stale Claude definitions by rerunning `set` with saved model and effort -in the same scope. This updates owned, unedited definitions while preserving -other preferences and unrelated/edited files; see README upgrades. Restart after -setup/regeneration. Bundled defaults need no setup. UG loads the bundled Claude -roles as `ug-smart-router:` only for a routed launch. Smart routing can -replace the requested model and role, so always include role instructions in -the delegated prompt and use runtime evidence to identify the model that ran. +Use runtime evidence to identify the model that ran; submitting a delegation +alone does not prove that its routing hook executed. diff --git a/skills/orchestrate/agents/explorer.md b/skills/orchestrate/agents/explorer.md index 4d68fe44b..d030391a6 100644 --- a/skills/orchestrate/agents/explorer.md +++ b/skills/orchestrate/agents/explorer.md @@ -1,7 +1,7 @@ --- name: explorer description: Map code, callers, existing patterns, and tests for a bounded investigation without editing files. -model: sonnet +model: inherit --- Follow the supervisor's bounded task contract. Inspect source and callers; report diff --git a/skills/orchestrate/agents/researcher.md b/skills/orchestrate/agents/researcher.md index 5c4783e84..b157d616f 100644 --- a/skills/orchestrate/agents/researcher.md +++ b/skills/orchestrate/agents/researcher.md @@ -1,7 +1,7 @@ --- name: researcher description: Verify external documentation and API facts against primary sources without editing files. -model: sonnet +model: inherit --- Follow the supervisor's bounded task. Cite source links; distinguish observations diff --git a/skills/orchestrate/agents/reviewer.md b/skills/orchestrate/agents/reviewer.md index 14d9e3eaf..bb224bac4 100644 --- a/skills/orchestrate/agents/reviewer.md +++ b/skills/orchestrate/agents/reviewer.md @@ -1,7 +1,7 @@ --- name: reviewer description: Independently inspect the resulting implementation for correctness, regressions, security, and missing tests. -model: sonnet +model: inherit --- Follow the supervisor's bounded task contract. Read the changed code and relevant diff --git a/skills/orchestrate/agents/tester.md b/skills/orchestrate/agents/tester.md index b9595e391..6a0b294c2 100644 --- a/skills/orchestrate/agents/tester.md +++ b/skills/orchestrate/agents/tester.md @@ -1,7 +1,7 @@ --- name: tester description: Independently execute acceptance checks and diagnose failures; edit tests only when assigned. -model: sonnet +model: inherit --- Follow the supervisor's bounded task contract. Run the requested checks and report diff --git a/skills/orchestrate/agents/worker.md b/skills/orchestrate/agents/worker.md index 3cb241176..7d95005a6 100644 --- a/skills/orchestrate/agents/worker.md +++ b/skills/orchestrate/agents/worker.md @@ -1,7 +1,7 @@ --- name: worker description: Implement one bounded change in files explicitly assigned by the supervisor. -model: sonnet +model: inherit --- Follow the supervisor's bounded task contract and file ownership. Preserve other diff --git a/skills/orchestrate/scripts/configure.py b/skills/orchestrate/scripts/configure.py deleted file mode 100644 index 067a84d8d..000000000 --- a/skills/orchestrate/scripts/configure.py +++ /dev/null @@ -1,438 +0,0 @@ -#!/usr/bin/env python3 -"""Resolve shared model preferences and manage owned Claude agent definitions.""" - -import argparse -import hashlib -import json -import os -import re -import tempfile -import tomllib -from contextlib import ExitStack, contextmanager -from pathlib import Path - -from ucode.codex_config import ( - DEFAULT_CODEX_CONFIG_PATH, - codex_config_precedence_paths, - codex_managed_config_path, -) -from ucode.os_compatibility.file_lock_cross_os import acquire_exclusive_file_lock, release_file_lock -from ucode.smart_routing.orchestrator import require_enabled - -PACKAGE = Path(__file__).resolve().parents[1] -ROLES = ("explorer", "researcher", "worker", "tester", "reviewer") -EFFORTS = { - "claude": ("low", "medium", "high", "xhigh", "max"), - "codex": ("none", "minimal", "low", "medium", "high", "xhigh", "max"), -} -_CODEX_GATEWAY_PREFIX = "system.ai." -_CODEX_NATIVE_GPT_VERSION = re.compile(r"^(gpt-\d+)\.(\d+)(?=-|$)") -_CODEX_GATEWAY_GPT_VERSION = re.compile(r"^(gpt-\d+)-(\d+)(?=-|$)") - - -def check_path(path): - for part in (path, *path.parents): - if part.is_symlink(): - raise ValueError(f"Refusing symlink: {part}") - if path.exists() and not path.is_file(): - raise ValueError(f"Not a regular file: {path}") - - -def validate_choice(harness, role, choice): - if not isinstance(choice, dict) or set(choice) - {"model", "effort"}: - raise ValueError(f"Invalid choice for {harness}/{role}") - model = choice.get("model") - if ( - not isinstance(model, str) - or not model - or any(character.isspace() or not character.isprintable() for character in model) - ): - raise ValueError(f"Invalid model for {harness}/{role}") - if choice.get("effort") is not None and choice["effort"] not in EFFORTS[harness]: - raise ValueError(f"Invalid effort for {harness}/{role}: expected {EFFORTS[harness]}") - - -def codex_model_candidates(model): - if "/" in model or ":" in model: - return [model] - if model.startswith(_CODEX_GATEWAY_PREFIX): - gateway_slug = model.removeprefix(_CODEX_GATEWAY_PREFIX) - native = _CODEX_GATEWAY_GPT_VERSION.sub(r"\1.\2", gateway_slug, count=1) - return [model, native] if native else [model] - gateway_slug = _CODEX_NATIVE_GPT_VERSION.sub(r"\1-\2", model, count=1) - return [model, _CODEX_GATEWAY_PREFIX + gateway_slug] - - -def active_codex_catalog_path(): - for config_path in codex_config_precedence_paths( - codex_managed_config_path(), DEFAULT_CODEX_CONFIG_PATH - ): - if not config_path.exists(): - continue - try: - with config_path.open("rb") as stream: - configured = tomllib.load(stream).get("model_catalog_json") - except (OSError, TypeError, ValueError) as error: - raise ValueError(f"Invalid Codex configuration: {config_path}") from error - if configured is None: - continue - if not isinstance(configured, str) or not configured: - raise ValueError(f"Invalid model_catalog_json in {config_path}") - path = Path(configured).expanduser() - return (path if path.is_absolute() else config_path.parent / path).resolve() - return None - - -def read_codex_catalog_slugs(path): - try: - catalog = json.loads(path.read_text()) - except (OSError, json.JSONDecodeError, UnicodeError) as error: - raise ValueError(f"Invalid Codex model catalog: {path}") from error - models = catalog.get("models") if isinstance(catalog, dict) else None - if ( - not isinstance(models, list) - or not models - or any( - not isinstance(model, dict) or not isinstance(model.get("slug"), str) - for model in models - ) - ): - raise ValueError(f"Invalid Codex model catalog: {path}") - return {model["slug"] for model in models} - - -def read_config(path): - check_path(path) - try: - data = json.loads(path.read_text()) if path.exists() else {} - except (json.JSONDecodeError, UnicodeError) as error: - raise ValueError( - f"Invalid JSON configuration: {path}; repair or restore it before retrying" - ) from error - if not isinstance(data, dict) or set(data) - {"claude", "codex", "_generated"}: - raise ValueError(f"Invalid configuration keys: {path}") - return data - - -def validate_ownership(data, path): - receipts = data.get("_generated", {}) - if not isinstance(receipts, dict) or set(receipts) - set(ROLES): - raise ValueError(f"Invalid ownership record: {path}") - if any( - not isinstance(receipt, str) or not re.fullmatch(r"[0-9a-f]{64}", receipt) - for receipt in receipts.values() - ): - raise ValueError(f"Invalid ownership hash: {path}") - - -def load_config(path, *, validate_choices=True, harness=None): - data = read_config(path) - for selected_harness in (harness,) if harness is not None else ("claude", "codex"): - roles = data.get(selected_harness, {}) - if not isinstance(roles, dict) or set(roles) - set(ROLES): - raise ValueError(f"Invalid {selected_harness} roles: {path}") - if validate_choices: - for role, choice in roles.items(): - validate_choice(selected_harness, role, choice) - if harness != "codex": - validate_ownership(data, path) - return data - - -def scope_paths(project): - if project is not None: - project = project.expanduser().resolve() - if not project.is_dir(): - raise ValueError(f"Project directory does not exist: {project}") - return project / ".model-orchestrator.json", project / ".claude" / "agents" - config_root = ( - Path(os.environ.get("XDG_CONFIG_HOME", str(Path.home() / ".config"))).expanduser().resolve() - ) - claude_root = ( - Path(os.environ.get("CLAUDE_CONFIG_DIR", str(Path.home() / ".claude"))) - .expanduser() - .resolve() - ) - return config_root / "model-orchestrator" / "config.json", claude_root / "agents" - - -def agent_name(role, project): - return f"model-orchestrator-custom-{'project' if project is not None else 'user'}-{role}" - - -def agent_content(role, choice, project): - text = (PACKAGE / "agents" / f"{role}.md").read_text() - text = re.sub(r"^name: .+$", f"name: {agent_name(role, project)}", text, count=1, flags=re.M) - model = "model: " + json.dumps(choice["model"]) - if choice.get("effort") is not None: - model += "\neffort: " + choice["effort"] - return re.sub(r"^model: .+$", lambda _: model, text, count=1, flags=re.M).encode() - - -def digest(content): - return hashlib.sha256(content).hexdigest() - - -def replace_file(path, content): - if content is None: - path.unlink(missing_ok=True) - return - path.parent.mkdir(parents=True, exist_ok=True) - fd, temporary = tempfile.mkstemp(prefix=".model-orchestrator-", dir=path.parent) - try: - with os.fdopen(fd, "wb") as stream: - stream.write(content) - stream.flush() - os.fsync(stream.fileno()) - os.replace(temporary, path) - finally: - Path(temporary).unlink(missing_ok=True) - - -@contextmanager -def configuration_lock(path, *, read_only_lock_file=False): - check_path(path) - path.parent.mkdir(parents=True, exist_ok=True) - lock = path.with_name(path.name + ".lock") - if lock.is_dir() and not lock.is_symlink(): - raise ValueError(f"Legacy lock directory: {lock}; remove only after its writer exits") - check_path(lock) - mode = "rb" if read_only_lock_file and lock.exists() else "a+b" - with lock.open(mode) as stream: - acquire_exclusive_file_lock(stream) - try: - yield - finally: - release_file_lock(stream) - - -def transaction_paths(project): - config_path, agent_dir = scope_paths(project) - journal = config_path.with_name(config_path.name + ".transaction.json") - paths = {"config": config_path} - paths.update({role: agent_dir / f"{agent_name(role, project)}.md" for role in ROLES}) - return journal, paths - - -def recovery_entry(before, after): - return { - "before": before.hex() if before is not None else None, - "after": after.hex() if after is not None else None, - } - - -def recover_scope(project): - journal, paths = transaction_paths(project) - check_path(journal) - if not journal.exists(): - return - entries = json.loads(journal.read_text()) - if not isinstance(entries, dict) or "config" not in entries or set(entries) - set(paths): - raise ValueError(f"Invalid recovery journal: {journal}") - restores = [] - for name, entry in entries.items(): - if not isinstance(entry, dict) or set(entry) != {"before", "after"}: - raise ValueError(f"Invalid recovery entry: {journal}") - if any(value is not None and not isinstance(value, str) for value in entry.values()): - raise ValueError(f"Invalid recovery content: {journal}") - before = bytes.fromhex(entry["before"]) if entry["before"] is not None else None - after = bytes.fromhex(entry["after"]) if entry["after"] is not None else None - target = paths[name] - check_path(target) - existing = target.read_bytes() if target.exists() else None - if existing not in (before, after): - raise ValueError(f"Preserving edited file during recovery: {target}") - if existing != before: - restores.append((target, before)) - for target, content in reversed(restores): - replace_file(target, content) - journal.unlink() - - -def update_scope(project, previous, desired, *, manage_claude=True): - config_path, agent_dir = scope_paths(project) - changes = {} - receipts = {} - for role in ROLES if manage_claude else (): - path = agent_dir / f"{agent_name(role, project)}.md" - owned = previous.get("_generated", {}).get(role) - choice = desired.get("claude", {}).get(role) - if not owned and choice is None: - continue - check_path(path) - existing = path.read_bytes() if path.exists() else None - if existing is not None and (not owned or digest(existing) != owned): - raise ValueError(f"Preserving unowned or edited agent: {path}") - content = agent_content(role, choice, project) if choice else None - if content is not None: - receipts[role] = digest(content) - if existing != content: - changes[path] = content - if manage_claude: - desired = {key: value for key, value in desired.items() if key != "_generated"} - if receipts: - desired["_generated"] = receipts - check_path(config_path) - changes[config_path] = (json.dumps(desired, indent=2) + "\n").encode() if desired else None - before = {path: path.read_bytes() if path.exists() else None for path in changes} - changes = {path: content for path, content in changes.items() if before[path] != content} - if not changes: - return [] - journal, paths = transaction_paths(project) - check_path(journal) - if journal.exists(): - raise ValueError(f"Pending recovery journal: {journal}; run show before updating") - entries = { - name: recovery_entry(before[path], changes[path]) - for name, path in paths.items() - if path in changes - } - if "config" not in entries: - entries["config"] = recovery_entry(before[config_path], before[config_path]) - replace_file(journal, (json.dumps(entries) + "\n").encode()) - try: - for path, content in changes.items(): - replace_file(path, content) - journal.unlink() - except OSError: - recover_scope(project) - raise - return [str(path) for path in changes] - - -def resolve(harness, project): - require_enabled() - with ExitStack() as locks: - scopes = [] - for scope in [None, project] if project is not None else [None]: - config_path = scope_paths(scope)[0] - journal = transaction_paths(scope)[0] - check_path(journal) - if config_path.exists() or journal.exists(): - locks.enter_context(configuration_lock(config_path, read_only_lock_file=True)) - recover_scope(scope) - scopes.append( - (scope, load_config(config_path, validate_choices=False, harness=harness)) - ) - catalog_path = active_codex_catalog_path() if harness == "codex" else None - return resolve_models(harness, scopes, catalog_path) - - -def resolve_models(harness, scopes, catalog_path=None): - catalog_slugs = read_codex_catalog_slugs(catalog_path) if catalog_path is not None else None - result = {} - for role in ROLES: - choice = ( - {"model": "gpt-5.6-luna", "effort": "max"} - if harness == "codex" - else {"model": "sonnet", "effort": None} - ) - configured = False - subagent_type = f"ug-smart-router:{role}" - for scope, data in reversed(scopes): - if role not in data.get(harness, {}): - continue - validate_choice(harness, role, data[harness][role]) - choice = {"effort": None, **data[harness][role]} - configured = True - if harness == "claude": - subagent_type = agent_name(role, scope) - path = scope_paths(scope)[1] / f"{subagent_type}.md" - check_path(path) - expected = agent_content(role, choice, scope) - if ( - not path.exists() - or path.read_bytes() != expected - or data.get("_generated", {}).get(role) != digest(expected) - ): - raise ValueError( - f"Missing/stale Claude agent; run set for {role} again: {path}" - ) - break - result[role] = dict(choice) - if harness == "claude": - result[role]["subagent_type"] = subagent_type - else: - result[role]["reasoning_effort"] = result[role].pop("effort") - result[role]["allow_inherited_fallback"] = not configured - candidates = codex_model_candidates(result[role]["model"]) - if catalog_slugs is not None and len(candidates) > 1: - # A missing catalog entry is not an invalid preference. Preserve - # the fallback policy while native spawn establishes availability. - result[role]["model"] = next( - (candidate for candidate in candidates if candidate in catalog_slugs), - result[role]["model"], - ) - if harness == "claude" and os.environ.get("CLAUDE_CODE_SUBAGENT_MODEL_FORCE", "").lower() in ( - "1", - "true", - ): - forced = os.environ.get("CLAUDE_CODE_SUBAGENT_MODEL") - if not forced or forced == "inherit": - raise ValueError( - "CLAUDE_CODE_SUBAGENT_MODEL_FORCE selects the parent model; role models cannot be resolved" - ) - if any(choice["model"] != forced for choice in result.values()): - raise ValueError( - "CLAUDE_CODE_SUBAGENT_MODEL_FORCE conflicts with the configured role models" - ) - return result - - -def main(): - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("action", choices=("show", "set", "unconfigure")) - parser.add_argument("--harness", choices=tuple(EFFORTS)) - parser.add_argument("--role", choices=ROLES) - parser.add_argument("--model") - parser.add_argument("--effort") - scope = parser.add_mutually_exclusive_group(required=True) - scope.add_argument("--project", type=Path) - scope.add_argument("--user", action="store_true") - args = parser.parse_args() - if args.action == "unconfigure" and args.harness: - parser.error("unconfigure removes the selected scope; omit --harness") - if args.action != "unconfigure" and not args.harness: - parser.error("show/set requires --harness") - if args.action == "set" and (not args.role or not args.model): - parser.error("set requires --role and --model") - if args.action != "set" and any((args.role, args.model, args.effort)): - parser.error("--role, --model, and --effort require set") - result: dict - try: - if args.action == "show": - result = resolve(args.harness, args.project) - else: - path = scope_paths(args.project)[0] - with configuration_lock(path): - recover_scope(args.project) - if args.action == "unconfigure": - previous = read_config(path) - validate_ownership(previous, path) - else: - previous = load_config( - path, validate_choices=args.harness != "codex", harness=args.harness - ) - desired: dict = json.loads(json.dumps(previous)) if args.action == "set" else {} - if args.action == "set": - desired.setdefault(args.harness, {})[args.role] = { - "model": args.model, - "effort": args.effort, - } - validate_choice(args.harness, args.role, desired[args.harness][args.role]) - result = { - "changed": update_scope( - args.project, previous, desired, manage_claude=args.harness != "codex" - ) - } - if args.harness == "claude" and result["changed"]: - result["next"] = ( - "Restart Claude Code after initial setup; run show to verify the resolved map." - ) - print(json.dumps(result, indent=2)) - except (OSError, ValueError) as error: - parser.exit(1, f"{error}\n") - - -if __name__ == "__main__": - main() diff --git a/src/ucode/smart_routing/orchestrator.py b/src/ucode/smart_routing/orchestrator.py index 64cd1c711..a538fbee0 100644 --- a/src/ucode/smart_routing/orchestrator.py +++ b/src/ucode/smart_routing/orchestrator.py @@ -2,6 +2,7 @@ from __future__ import annotations +import argparse import json import os import shlex @@ -22,7 +23,7 @@ "This supersedes any earlier model-orchestrator workflow: do not start new automatic " "delegation or fall back to orchestrator role models. Explicit user requests for subagents " "still use native tools and normal harness model selection, without the orchestrator " - "or its model-resolution helper; keep routing off. Otherwise continue the task in the root. " + "or its routing check; keep routing off. Otherwise continue the task in the root. " "Collect results from children already running." ) @@ -150,7 +151,20 @@ def hook_output(payload: object) -> dict | None: return {"hookSpecificOutput": {"hookEventName": event, "additionalContext": context}} -def main() -> None: +def main(argv: list[str] | None = None) -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--check", + action="store_true", + help="Exit successfully only in an enabled smart-routing session.", + ) + if parser.parse_args(argv).check: + try: + require_enabled() + except ValueError as exc: + parser.exit(1, f"{exc}\n") + return + try: payload = json.load(sys.stdin) except (OSError, UnicodeError, ValueError): diff --git a/tests/README.md b/tests/README.md index 50c97de01..98cbf986b 100644 --- a/tests/README.md +++ b/tests/README.md @@ -133,11 +133,13 @@ PowerShell execution. `test_orchestrator.py` covers shared routing state, off/on transitions, suppression of retained skills outside eligible sessions, root prompts and compaction, -user-hook preservation, and bundled Claude roles. `test_orchestrator_config.py` -ports the plugin's preference, ownership, locking, and interrupted-write recovery -coverage and checks UG catalog precedence. Real subprocess tests verify that preference -reads, writes, and removal wait for an existing writer, preserve files while waiting, -and complete after the lock is released. `test_orchestrator_legacy_plugins.py` +user-hook preservation, and bundled Claude roles. Subprocess checks verify that +`--check` rejects ineligible sessions and leaves legacy preference files untouched, +even when they are malformed. The installed skill's check command follows live +toggles using the launching interpreter in `test_smart_router.py`. Routing checks +cover Claude/Codex role contracts without model preferences. Separate preference +editing, locking, and recovery are no longer part of the workflow. +`test_orchestrator_legacy_plugins.py` and the Claude/Codex launcher tests check native per-launch overrides that disable legacy marketplace registrations with routing on or off, including Codex's app-server and remote TUI. They check config discovery, unrelated-plugin and hook diff --git a/tests/integration/README.md b/tests/integration/README.md index 5d81bfefb..b8766477f 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -288,9 +288,11 @@ Collapsed terminal output is allowed; the answer need not repeat the CLI's exact Each following child still verifies whether a routing decision occurred. Their off-phase child is an explicit user-requested delegation; these journeys do not establish automatic orchestration behavior. Shared on/off -state, root-only activation, compaction, retained skills, and role preference -compatibility are covered in `../test_orchestrator.py` and -`../test_orchestrator_config.py`. `../test_orchestrator_legacy_plugins.py` and +state, root-only activation, compaction, retained skills, role contracts, and +isolation from legacy preferences are covered in `../test_orchestrator.py`. +`../test_smart_router.py` executes the installed routing-check command across +session toggles. Separate preference editing, locking, and recovery are not part +of the workflow. `../test_orchestrator_legacy_plugins.py` and launcher component tests check per-launch suppression of installed legacy plugins, including non-routed launches, nested project config and `--cd`, and preservation of saved settings and unrelated plugins/hooks. Native config protocol and timeout diff --git a/tests/test_orchestrator.py b/tests/test_orchestrator.py index 148827983..96b9071bc 100644 --- a/tests/test_orchestrator.py +++ b/tests/test_orchestrator.py @@ -13,7 +13,7 @@ from typer.testing import CliRunner from ucode import cli, skills -from ucode.smart_routing import orchestrator, routing, session_env, v2 +from ucode.smart_routing import codex_routing, orchestrator, routing, session_env, v2 @pytest.fixture @@ -64,6 +64,53 @@ def test_retained_skill_cannot_enable_orchestration(routed_session, monkeypatch, assert not orchestrator.enabled() with pytest.raises(ValueError, match="do not start new automatic delegation"): orchestrator.require_enabled() + result = subprocess.run( + [sys.executable, "-m", orchestrator.HOOK_MODULE, "--check"], + stdin=subprocess.DEVNULL, + capture_output=True, + text=True, + timeout=20, + ) + assert result.returncode == 1 + assert result.stdout == "" + assert result.stderr == orchestrator.DISABLED_CONTEXT + "\n" + + +def test_check_does_not_read_or_modify_legacy_preferences(routed_session, tmp_path): + legacy_files = { + ".model-orchestrator.json": b"{invalid project preferences", + ".model-orchestrator.json.transaction.json": b"{unfinished update", + ".config/model-orchestrator/config.json": b"{invalid user preferences", + ".config/model-orchestrator/config.json.lock": b"existing lock file", + ".claude/agents/model-orchestrator-custom-project-reviewer.md": b"user-edited agent", + ".codex/config.toml": b"[invalid catalog config", + } + for relative, content in legacy_files.items(): + path = tmp_path / relative + path.parent.mkdir(parents=True, exist_ok=True) + path.write_bytes(content) + before = {path: path.read_bytes() for path in tmp_path.rglob("*") if path.is_file()} + + result = subprocess.run( + [sys.executable, "-m", orchestrator.HOOK_MODULE, "--check"], + cwd=tmp_path, + env={ + **os.environ, + "HOME": str(tmp_path), + "USERPROFILE": str(tmp_path), + "XDG_CONFIG_HOME": str(tmp_path / ".config"), + "CLAUDE_CONFIG_DIR": str(tmp_path / ".claude"), + "CODEX_HOME": str(tmp_path / ".codex"), + }, + stdin=subprocess.DEVNULL, + capture_output=True, + text=True, + timeout=20, + ) + + assert result.returncode == 0, result.stderr + assert result.stdout == result.stderr == "" + assert {path: path.read_bytes() for path in tmp_path.rglob("*") if path.is_file()} == before @pytest.mark.parametrize( @@ -126,7 +173,7 @@ def test_only_root_prompt_and_compaction_load_workflow(routed_session, payload, @pytest.mark.parametrize("payload", ["", "{", "null", "[]", '"text"', "{}"]) def test_hook_entry_point_ignores_invalid_payload(monkeypatch, capsys, payload): monkeypatch.setattr(sys, "stdin", io.StringIO(payload)) - orchestrator.main() + orchestrator.main([]) assert capsys.readouterr().out == "" @@ -231,3 +278,24 @@ def test_role_contract_survives_claude_model_routing(monkeypatch): assert updated["prompt"] == contract assert updated["subagent_type"].startswith("ug-smart-router:ucode-route-") assert "model" not in updated + + +def test_codex_routes_role_without_model_preferences(monkeypatch): + # Keep the routing path real; replace only the external selection request. + monkeypatch.setattr( + codex_routing, + "request_routing_decision", + lambda *_args, **_kwargs: ( + routing.RoutingDecision(model="system.ai.gpt-5-6-sol", raw_model="gpt-5-6-sol"), + None, + ), + ) + contract = "Act as the reviewer. Inspect this diff without editing. Report concrete bugs." + task = {"task_name": "reviewer", "message": contract, "fork_turns": "none"} + result = codex_routing.route_pre_tool_use( + {"tool_name": "collaboration.spawn_agent", "tool_input": task}, + workspace="https://example.com", + token="token", + available_models=["system.ai.gpt-5-6-sol"], + ) + assert result["hookSpecificOutput"]["updatedInput"] == {**task, "model": "gpt-5.6-sol"} diff --git a/tests/test_orchestrator_config.py b/tests/test_orchestrator_config.py deleted file mode 100644 index 2f02ff059..000000000 --- a/tests/test_orchestrator_config.py +++ /dev/null @@ -1,954 +0,0 @@ -"""Exercise real CLI configuration, ownership, and removal without model calls.""" - -import errno -import importlib.util -import json -import os -import shutil -import stat -import subprocess -import sys -import time -from pathlib import Path -from typing import Any - -import pytest - -from ucode.os_compatibility.file_lock_cross_os import acquire_exclusive_file_lock, release_file_lock -from ucode.smart_routing import session_env - -# Only replace the external machine config path; run the real helper and file operations. -CONFIG_CLI = """ -import runpy -import sys -from pathlib import Path -from ucode import codex_config -script, managed, *arguments = sys.argv[1:] -codex_config.codex_managed_config_path = lambda: Path(managed) -sys.argv = [script, *arguments] -runpy.run_path(script, run_name="__main__") -""" - - -class ConfigHarness: - def __init__(self, root: Path, script: Path): - self.root = root - self.script = script - self.project = root / "project with spaces" - self.project.mkdir() - self.env = dict( - os.environ, - XDG_CONFIG_HOME=str(root / "config"), - CLAUDE_CONFIG_DIR=str(root / "claude"), - CODEX_HOME=str(root / "codex-home"), - ) - self.env.pop("CLAUDE_CODE_SUBAGENT_MODEL_FORCE", None) - self.env["ENABLE_SMART_ROUTING_V2"] = "1" - self.env.pop("ENABLE_SMART_ROUTING_SUBAGENT_ONLY", None) - session_env.start_session(self.env) - self.env.pop("ISAAC_LAUNCH_MODE", None) - - @property - def config(self): - return self.project / ".model-orchestrator.json" - - @property - def journal(self): - return self.config.with_name(self.config.name + ".transaction.json") - - @property - def user_config(self): - return Path(self.env["XDG_CONFIG_HOME"]) / "model-orchestrator/config.json" - - def cli(self, *args, user=False, ok=True) -> Any: - scope = ["--user"] if user else ["--project", str(self.project)] - result = subprocess.run( - [ - sys.executable, - "-c", - CONFIG_CLI, - str(self.script), - str(self.root / "managed.toml"), - *args, - *scope, - ], - env=self.env, - capture_output=True, - text=True, - timeout=20, - ) - assert (result.returncode == 0) == ok, result.stderr - return json.loads(result.stdout) if ok else result.stderr - - def set_model(self, harness="claude", role="worker", model="provider/custom-model", **kwargs): - return self.cli("set", "--harness", harness, "--role", role, "--model", model, **kwargs) - - def set_catalog(self, *models): - catalog = self.root / "codex-model-catalog.json" - catalog.write_text(json.dumps({"models": [{"slug": model} for model in models]})) - codex_home = Path(self.env["CODEX_HOME"]) - codex_home.mkdir(exist_ok=True) - (codex_home / "ucode.config.toml").write_text( - f"model_catalog_json = {json.dumps(str(catalog))}\n" - ) - return catalog - - def agent(self, role="worker", user=False): - root = Path(self.env["CLAUDE_CONFIG_DIR"]) if user else self.project / ".claude" - return ( - root / "agents" / f"model-orchestrator-custom-{'user' if user else 'project'}-{role}.md" - ) - - def interrupt(self, target, *arguments): - program = """ -import importlib.util -import os -from pathlib import Path -import sys - -script, target, *arguments = sys.argv[1:] -spec = importlib.util.spec_from_file_location("orchestrator_configure", script) -module = importlib.util.module_from_spec(spec) -spec.loader.exec_module(module) -original_replace = os.replace -original_unlink = os.unlink - -def replace(source, destination): - original_replace(source, destination) - if Path(destination) == Path(target): - os._exit(17) - -def unlink(destination, *args, **kwargs): - original_unlink(destination, *args, **kwargs) - if Path(destination) == Path(target): - os._exit(17) - -os.replace = replace -os.unlink = unlink -sys.argv = [script, *arguments] -module.main() -""" - result = subprocess.run( - [ - sys.executable, - "-c", - program, - str(self.script), - str(target), - *arguments, - "--project", - str(self.project), - ], - env=self.env, - capture_output=True, - text=True, - timeout=20, - ) - assert result.returncode == 17, result.stderr - - -@pytest.fixture -def config(tmp_path): - return ConfigHarness( - tmp_path, Path(__file__).parents[1] / "skills/orchestrate/scripts/configure.py" - ) - - -@pytest.fixture -def configure_module(config, monkeypatch): - spec = importlib.util.spec_from_file_location("orchestrator_configure", config.script) - assert spec is not None and spec.loader is not None - module = importlib.util.module_from_spec(spec) - spec.loader.exec_module(module) - monkeypatch.setattr(module, "codex_managed_config_path", lambda: config.root / "managed.toml") - for key in (session_env.SESSION_ENV_VAR, "ENABLE_SMART_ROUTING_V2"): - monkeypatch.setenv(key, config.env[key]) - return module - - -def test_defaults_need_no_files(config): - claude = config.cli("show", "--harness", "claude") - codex = config.cli("show", "--harness", "codex") - assert {choice["model"] for choice in claude.values()} == {"sonnet"} - assert claude["reviewer"]["subagent_type"] == "ug-smart-router:reviewer" - assert codex["worker"] == { - "model": "gpt-5.6-luna", - "reasoning_effort": "max", - "allow_inherited_fallback": True, - } - assert list(config.project.iterdir()) == [] - assert not Path(config.env["XDG_CONFIG_HOME"]).exists() - - -@pytest.mark.parametrize("harness", ["claude", "codex"]) -@pytest.mark.parametrize("configured", [False, True]) -@pytest.mark.parametrize("state", ["off", "no-session"]) -def test_show_never_authorizes_delegation_when_routing_is_off(config, harness, configured, state): - if configured: - config.set_model(harness=harness) - if state == "off": - Path(config.env[session_env.SESSION_ENV_VAR]).write_text( - '{"ENABLE_SMART_ROUTING_V2":"0","ENABLE_SMART_ROUTING_SUBAGENT_ONLY":"0"}' - ) - else: - config.env.pop(session_env.SESSION_ENV_VAR) - error = config.cli("show", "--harness", harness, ok=False) - assert "do not start new automatic delegation" in error - # Preferences can still be managed without enabling either feature. - config.set_model(harness=harness) - config.cli("unconfigure") - - -def test_codex_managed_catalog_precedes_ug_and_user_catalogs(config): - catalog = config.set_catalog("gpt-5.6-luna") - managed_catalog = config.root / "managed-models.json" - managed_catalog.write_text('{"models":[{"slug":"system.ai.gpt-5-6-luna"}]}') - (config.root / "managed.toml").write_text( - f"model_catalog_json = {json.dumps(str(managed_catalog))}\n" - ) - (Path(config.env["CODEX_HOME"]) / "config.toml").write_text( - f"model_catalog_json = {json.dumps(str(catalog))}\n" - ) - assert config.cli("show", "--harness", "codex")["worker"]["model"] == ("system.ai.gpt-5-6-luna") - - -@pytest.mark.parametrize( - "configured,equivalent", - [ - ("gpt-5.6-luna", "system.ai.gpt-5-6-luna"), - ("system.ai.gpt-5-6-luna", "gpt-5.6-luna"), - ("gpt-6-luna", "system.ai.gpt-6-luna"), - ("system.ai.gpt-6-sol", "gpt-6-sol"), - ("glm-5-3", "system.ai.glm-5-3"), - ("system.ai.deepseek-v4-1-flash", "deepseek-v4-1-flash"), - ], -) -def test_codex_catalog_aliases_resolve_to_an_exact_available_model(config, configured, equivalent): - config.set_model(harness="codex", model=configured) - config.set_catalog(equivalent) - worker = config.cli("show", "--harness", "codex")["worker"] - assert worker == { - "model": equivalent, - "reasoning_effort": None, - "allow_inherited_fallback": False, - } - - -def test_codex_catalog_prefers_the_configured_spelling(config): - config.set_model(harness="codex", model="gpt-5.6-luna") - config.set_catalog("system.ai.gpt-5-6-luna", "gpt-5.6-luna") - assert config.cli("show", "--harness", "codex")["worker"]["model"] == "gpt-5.6-luna" - - -def test_codex_full_model_id_is_not_rewritten(config): - config.set_model(harness="codex", model="provider/custom-model") - assert config.cli("show", "--harness", "codex")["worker"] == { - "model": "provider/custom-model", - "reasoning_effort": None, - "allow_inherited_fallback": False, - } - - -@pytest.mark.parametrize( - "model", - [ - "provider/custom-model", - "provider:custom-model", - "system.ai.provider/custom-model", - "system.ai.provider:custom-model", - ], -) -def test_codex_custom_model_id_bypasses_catalog_alias_matching(config, model): - config.set_catalog("system.ai.gpt-5-6-luna", "provider/custom-model", "provider:custom-model") - config.set_model(harness="codex", model=model) - assert config.cli("show", "--harness", "codex")["worker"]["model"] == model - - -@pytest.mark.parametrize( - "configured,other_version", - [ - ("gpt-6-luna", "gpt-5.6-luna"), - ("gpt-5.6-luna", "gpt-6-luna"), - ("gpt-6-sol", "gpt-5.6-sol"), - ("gpt-5.6-sol", "gpt-6-sol"), - ], -) -def test_codex_catalog_aliases_do_not_cross_model_versions(config, configured, other_version): - config.set_model(harness="codex", model=configured) - config.set_catalog(other_version) - assert config.cli("show", "--harness", "codex")["worker"] == { - "model": configured, - "reasoning_effort": None, - "allow_inherited_fallback": False, - } - - -@pytest.mark.parametrize("available", [True, False]) -def test_codex_catalog_preserves_bundled_fallback_eligibility(config, available): - model = "system.ai.gpt-5-6-luna" if available else "another-model" - config.set_catalog(model) - roles = config.cli("show", "--harness", "codex") - assert set(roles) == {"explorer", "researcher", "worker", "tester", "reviewer"} - for choice in roles.values(): - assert choice == { - "model": model if available else "gpt-5.6-luna", - "reasoning_effort": "max", - "allow_inherited_fallback": True, - } - - -def test_codex_unavailable_role_does_not_block_an_available_role(config): - config.set_model(harness="codex", role="explorer", model="missing-model") - config.set_model(harness="codex", role="reviewer", model="glm-5-3") - config.set_catalog("system.ai.glm-5-3") - roles = config.cli("show", "--harness", "codex") - assert roles["explorer"] == { - "model": "missing-model", - "reasoning_effort": None, - "allow_inherited_fallback": False, - } - assert roles["reviewer"] == { - "model": "system.ai.glm-5-3", - "reasoning_effort": None, - "allow_inherited_fallback": False, - } - assert roles["worker"] == { - "model": "gpt-5.6-luna", - "reasoning_effort": "max", - "allow_inherited_fallback": True, - } - - -def test_codex_catalog_rejects_missing_or_malformed_catalog(config): - missing = config.set_catalog("system.ai.gpt-5-6-luna") - missing.unlink() - assert str(missing) in config.cli("show", "--harness", "codex", ok=False) - missing.write_text("not json") - assert str(missing) in config.cli("show", "--harness", "codex", ok=False) - - -@pytest.mark.parametrize( - "contents", ["[]", "{}", '{"models": []}', '{"models": [null]}', '{"models": [{"slug": 5}]}'] -) -def test_codex_catalog_rejects_invalid_structure(config, contents): - catalog = config.set_catalog("system.ai.gpt-5-6-luna") - catalog.write_text(contents) - assert str(catalog) in config.cli("show", "--harness", "codex", ok=False) - - -def test_codex_catalog_falls_back_to_codex_config(config): - catalog = config.set_catalog("system.ai.gpt-5-6-luna") - codex_home = config.root / "codex-home" - (codex_home / "ucode.config.toml").unlink() - (codex_home / "config.toml").write_text(f"model_catalog_json = {json.dumps(str(catalog))}\n") - config.env["CODEX_HOME"] = str(codex_home) - assert config.cli("show", "--harness", "codex")["worker"]["model"] == "system.ai.gpt-5-6-luna" - - -def test_codex_ug_catalog_takes_precedence_over_user_config(config): - config.set_catalog("system.ai.gpt-5-6-luna") - codex_home = Path(config.env["CODEX_HOME"]) - codex_home.mkdir(exist_ok=True) - (codex_home / "config.toml").write_text('model_catalog_json = "missing.json"\n') - assert config.cli("show", "--harness", "codex")["worker"]["model"] == "system.ai.gpt-5-6-luna" - - -def test_codex_catalog_config_relative_path(config): - codex_home = Path(config.env["CODEX_HOME"]) - codex_home.mkdir(exist_ok=True) - (codex_home / "models.json").write_text('{"models": [{"slug": "system.ai.gpt-5-6-luna"}]}') - (codex_home / "config.toml").write_text('model_catalog_json = "models.json"\n') - assert config.cli("show", "--harness", "codex")["worker"]["model"] == "system.ai.gpt-5-6-luna" - - -@pytest.mark.parametrize( - "contents", ["model_catalog_json = 5", 'model_catalog_json = ""', "not toml"] -) -def test_codex_catalog_rejects_invalid_config(config, contents): - codex_home = Path(config.env["CODEX_HOME"]) - codex_home.mkdir(exist_ok=True) - config_path = codex_home / "config.toml" - config_path.write_text(contents) - assert str(config_path) in config.cli("show", "--harness", "codex", ok=False) - - -def test_codex_catalog_does_not_hide_invalid_preferences(config): - config.set_catalog("system.ai.gpt-5-6-luna") - config.config.write_text('{"codex": {"worker": {"model": ""}}}') - assert "Invalid model for codex/worker" in config.cli("show", "--harness", "codex", ok=False) - - -@pytest.mark.parametrize("catalog", [None, "system.ai.gpt-5-6-luna", "another-model"]) -@pytest.mark.parametrize("user", [False, True]) -@pytest.mark.parametrize("model", ["gpt-5.6-luna", "provider/custom-model"]) -def test_codex_inherited_fallback_preserves_explicit_model_choices(config, user, model, catalog): - if catalog is not None: - config.set_catalog(catalog) - config.set_model(harness="codex", model=model, user=user) - roles = config.cli("show", "--harness", "codex") - expected = catalog if model == "gpt-5.6-luna" and catalog == "system.ai.gpt-5-6-luna" else model - assert roles["worker"]["model"] == expected - assert roles["worker"]["allow_inherited_fallback"] is False - assert roles["reviewer"]["allow_inherited_fallback"] is True - config.cli("unconfigure", user=user) - assert config.cli("show", "--harness", "codex")["worker"]["allow_inherited_fallback"] is True - - -def test_codex_fallback_stays_disabled_when_project_override_reveals_user_choice(config): - config.set_model(harness="codex", model="user-model", user=True) - config.set_model(harness="codex", model="project-model") - worker = config.cli("show", "--harness", "codex")["worker"] - assert worker["model"] == "project-model" - assert worker["allow_inherited_fallback"] is False - config.cli("unconfigure") - worker = config.cli("show", "--harness", "codex")["worker"] - assert worker["model"] == "user-model" - assert worker["allow_inherited_fallback"] is False - - -@pytest.mark.parametrize("role", ["explorer", "researcher", "worker", "tester", "reviewer"]) -def test_bundled_claude_agents_use_sonnet(config, role): - agent = config.script.parents[1] / "agents" / f"{role}.md" - frontmatter = agent.read_text().split("---", 2)[1] - assert "model: sonnet" in frontmatter.splitlines() - - -def test_bundled_skill_inherits_supervisor_model(config): - skill = config.script.parents[1] / "SKILL.md" - frontmatter = skill.read_text().split("---", 2)[1] - assert "model: inherit" in frontmatter.splitlines() - - -def test_full_id_effort_idempotence_and_cleanup(config): - arguments = ( - "set", - "--harness", - "claude", - "--role", - "worker", - "--model", - "provider/model:#id", - "--effort", - "high", - ) - config.cli(*arguments) - agent = config.agent() - assert 'model: "provider/model:#id"\neffort: high' in agent.read_text() - stamp = agent.stat().st_mtime_ns - assert config.cli(*arguments)["changed"] == [] - assert agent.stat().st_mtime_ns == stamp - assert config.cli("show", "--harness", "claude")["worker"] == { - "model": "provider/model:#id", - "effort": "high", - "subagent_type": "model-orchestrator-custom-project-worker", - } - config.cli("unconfigure") - assert not agent.exists() - assert not config.config.exists() - assert not config.journal.exists() - - -def test_project_overrides_user_and_harnesses_stay_separate(config): - config.set_model(model="user-model", user=True) - config.set_model(model="project-model") - config.set_model(harness="codex", role="reviewer", model="other-model") - assert config.cli("show", "--harness", "claude")["worker"]["model"] == "project-model" - original = config.agent(user=True).read_bytes() - config.agent(user=True).write_text("Edited but shadowed by project override") - assert config.cli("show", "--harness", "claude")["worker"]["model"] == "project-model" - config.agent(user=True).write_bytes(original) - assert config.cli("show", "--harness", "codex")["reviewer"] == { - "model": "other-model", - "reasoning_effort": None, - "allow_inherited_fallback": False, - } - config.cli("unconfigure") - assert config.cli("show", "--harness", "claude")["worker"]["model"] == "user-model" - config.cli("unconfigure", "--harness", "claude", user=True, ok=False) - assert config.agent(user=True).exists() - - -def test_edited_and_unowned_agents_are_preserved(config): - config.set_model() - agent = config.agent() - original = agent.read_text() - agent.write_text(original + "User edit\n") - before = config.config.read_bytes() - assert "Preserving" in config.cli("unconfigure", ok=False) - assert config.config.read_bytes() == before - assert agent.read_text().endswith("User edit\n") - assert "Missing/stale" in config.cli("show", "--harness", "claude", ok=False) - agent.write_text(original) - config.cli("unconfigure") - agent.write_text("Unrelated file\n") - assert "Preserving" in config.set_model(ok=False) - assert agent.read_text() == "Unrelated file\n" - assert not config.config.exists() - - -@pytest.mark.parametrize("model", ["", "bad\nmodel", "bad model", "bad\x01model"]) -def test_invalid_models_do_not_write(config, model): - config.set_model(model=model, ok=False) - assert not config.config.exists() - assert not config.agent().exists() - - -def test_invalid_effort_does_not_write(config): - config.cli( - "set", - "--harness", - "claude", - "--role", - "worker", - "--model", - "sonnet", - "--effort", - "unsupported-effort", - ok=False, - ) - assert not config.config.exists() - assert not config.agent().exists() - - -@pytest.mark.parametrize("location", ["project", "XDG_CONFIG_HOME", "CLAUDE_CONFIG_DIR"]) -def test_symlinked_scope_roots_are_supported(config, location): - alias = config.root / "alias" - if location == "project": - alias.symlink_to(config.project, target_is_directory=True) - config.project = alias - else: - target = Path(config.env[location]) - target.mkdir() - alias.symlink_to(target, target_is_directory=True) - config.env[location] = str(alias) - user = location != "project" - config.set_model(user=user) - assert ( - config.cli("show", "--harness", "claude", user=user)["worker"]["model"] - == "provider/custom-model" - ) - config.cli("unconfigure", user=user) - assert not config.agent(user=user).exists() - - -@pytest.mark.parametrize( - "location", - [ - ".claude", - ".model-orchestrator.json", - ".model-orchestrator.json.lock", - ".model-orchestrator.json.transaction.json", - ], -) -def test_managed_symlinks_are_rejected(config, location): - outside = config.root / "outside" - if location == ".claude": - outside.mkdir() - else: - outside.write_text("Untouched") - (config.project / location).symlink_to(outside) - assert "symlink" in config.set_model(ok=False) - if outside.is_dir(): - assert list(outside.iterdir()) == [] - else: - assert outside.read_text() == "Untouched" - - -@pytest.mark.parametrize( - "data,harness", - [ - ([], "claude"), - ({"unknown": {}}, "claude"), - ({"codex": {"unknown": {}}}, "codex"), - ({"_generated": {"worker": "bad"}}, "claude"), - ], -) -def test_malformed_configuration_is_rejected(config, data, harness): - config.config.write_text(json.dumps(data)) - config.cli("show", "--harness", harness, ok=False) - - -def test_forced_model_policy_is_preserved(config): - config.env.update(CLAUDE_CODE_SUBAGENT_MODEL="opus", CLAUDE_CODE_SUBAGENT_MODEL_FORCE="1") - assert "conflicts" in config.cli("show", "--harness", "claude", ok=False) - config.env.pop("CLAUDE_CODE_SUBAGENT_MODEL") - assert "parent model" in config.cli("show", "--harness", "claude", ok=False) - config.cli("show", "--harness", "codex") - - -def test_codex_update_preserves_edited_claude_agents(config): - config.set_model() - config.agent().write_text("User customization\n") - receipts = json.loads(config.config.read_text())["_generated"] - config.set_model(harness="codex", model="codex-model") - assert config.agent().read_text() == "User customization\n" - assert json.loads(config.config.read_text())["_generated"] == receipts - assert config.cli("show", "--harness", "codex")["worker"]["model"] == "codex-model" - assert "Preserving" in config.cli("unconfigure", ok=False) - - -@pytest.mark.parametrize( - "section,value", - [ - ("claude", {"worker": {"model": "invalid model"}}), - ("claude", []), - ("claude", {"unknown": None}), - ("_generated", {"worker": "bad"}), - ("_generated", []), - ("_generated", None), - ], -) -def test_codex_update_preserves_malformed_claude_state(config, section, value): - config.set_model() - agent_before = config.agent().read_bytes() - data = json.loads(config.config.read_text()) - data[section] = value - config.config.write_text(json.dumps(data)) - config.set_model(harness="codex", model="codex-model") - updated = json.loads(config.config.read_text()) - assert updated["claude"] == data["claude"] - assert updated["_generated"] == data["_generated"] - assert config.agent().read_bytes() == agent_before - assert config.cli("show", "--harness", "codex")["worker"]["model"] == "codex-model" - - -def test_codex_set_repairs_only_the_selected_role(config): - config.config.write_text(json.dumps({"codex": {"worker": {"model": 4}, "reviewer": []}})) - config.set_model(harness="codex", model="codex-model") - updated = json.loads(config.config.read_text()) - assert updated["codex"] == {"worker": {"model": "codex-model", "effort": None}, "reviewer": []} - - -@pytest.mark.parametrize( - "section,value", - [ - ("claude", {"worker": {"model": 4}}), - ("claude", []), - ("claude", {"unknown": {}}), - ("codex", {"worker": {"model": "invalid model"}}), - ("codex", []), - ], -) -def test_unconfigure_uses_ownership_receipts_despite_invalid_roles(config, section, value): - config.set_model() - unrelated = config.agent(role="reviewer") - unrelated.write_text("Unrelated file\n") - data = json.loads(config.config.read_text()) - data[section] = value - config.config.write_text(json.dumps(data)) - config.cli("unconfigure") - assert not config.config.exists() - assert not config.agent().exists() - assert unrelated.read_text() == "Unrelated file\n" - - -@pytest.mark.parametrize("receipts", [{"worker": "bad"}, [], None]) -def test_unconfigure_preserves_files_when_ownership_is_invalid(config, receipts): - config.set_model() - agent_before = config.agent().read_bytes() - data = json.loads(config.config.read_text()) - data["_generated"] = receipts - config.config.write_text(json.dumps(data)) - config_before = config.config.read_bytes() - assert "ownership" in config.cli("unconfigure", ok=False) - assert config.config.read_bytes() == config_before - assert config.agent().read_bytes() == agent_before - - -@pytest.mark.parametrize("action", ["set", "unconfigure"]) -def test_corrupt_json_updates_and_cleanup_preserve_files(config, action): - config.set_model() - agent_before = config.agent().read_bytes() - config.config.write_text("{invalid json") - if action == "set": - error = config.set_model(harness="codex", ok=False) - else: - error = config.cli("unconfigure", ok=False) - assert "repair or restore" in error - assert config.config.read_text() == "{invalid json" - assert config.agent().read_bytes() == agent_before - - -@pytest.mark.parametrize("user", [False, True]) -@pytest.mark.parametrize("edited", [False, True]) -@pytest.mark.parametrize("role", ["explorer", "researcher", "worker", "tester", "reviewer"]) -def test_template_upgrade_refreshes_only_unedited_owned_agents(config, user, edited, role): - package = config.root / "plugin" - shutil.copytree(config.script.parents[1], package) - config.script = package / "scripts/configure.py" - arguments = ( - "set", - "--harness", - "claude", - "--role", - role, - "--model", - "provider/custom-model", - "--effort", - "high", - ) - config.cli(*arguments, user=user) - config_path = config.user_config if user else config.config - config_before = config_path.read_bytes() - template = package / "agents" / f"{role}.md" - template.chmod(template.stat().st_mode | stat.S_IWUSR) - template.write_text(template.read_text() + "\nUpdated role instructions.\n") - agent = config.agent(role=role, user=user) - if edited: - agent.write_text(agent.read_text() + "User edit\n") - assert "Missing/stale" in config.cli("show", "--harness", "claude", user=user, ok=False) - if edited: - assert "Preserving" in config.cli(*arguments, user=user, ok=False) - assert config_path.read_bytes() == config_before - assert agent.read_text().endswith("User edit\n") - else: - config.cli(*arguments, user=user) - assert agent.read_text().endswith("Updated role instructions.\n") - assert ( - json.loads(config_path.read_text())["_generated"] - != json.loads(config_before)["_generated"] - ) - resolved = config.cli("show", "--harness", "claude", user=user)[role] - assert resolved["model"] == "provider/custom-model" - assert resolved["effort"] == "high" - - -@pytest.mark.parametrize( - "choice", - [{"model": "invalid model"}, {"model": 4}, {"model": "sonnet", "effort": "invalid"}, []], -) -def test_invalid_shadowed_role_is_ignored_but_invalid_fallback_is_rejected(config, choice): - config.set_model(model="user-model", user=True) - config.set_model(model="project-model") - data = json.loads(config.user_config.read_text()) - data["claude"]["worker"] = choice - config.user_config.write_text(json.dumps(data)) - assert config.cli("show", "--harness", "claude")["worker"]["model"] == "project-model" - config.cli("show", "--harness", "claude", user=True, ok=False) - config.cli("show", "--harness", "codex") - - -def test_corrupt_user_json_is_not_hidden_by_project_overrides(config): - config.set_model(user=True) - config.set_model(model="project-model") - config.user_config.write_text("{invalid json") - config.cli("show", "--harness", "claude", ok=False) - - -@pytest.mark.parametrize("action", ["show", "set", "unconfigure"]) -def test_preference_operations_wait_for_existing_writer(config, action): - config.cli( - "set", "--harness", "claude", "--role", "worker", "--model", "sonnet", "--effort", "xhigh" - ) - before = config.config.read_bytes() - agent_before = config.agent().read_bytes() - lock = config.config.with_name(config.config.name + ".lock") - ready = config.root / "preference-operation-ready" - arguments = [action] if action == "unconfigure" else [action, "--harness", "claude"] - if action == "set": - arguments += ["--role", "reviewer", "--model", "sonnet"] - # Finish routing setup before timing the wait for the configuration lock. - program = """ -import runpy -import sys -from pathlib import Path -script, ready, *arguments = sys.argv[1:] -module = runpy.run_path(script) -module["require_enabled"]() -sys.argv = [script, *arguments] -Path(ready).touch() -module["main"]() -""" - process = None - try: - with lock.open("a+b") as stream: - acquire_exclusive_file_lock(stream) - try: - process = subprocess.Popen( - [ - sys.executable, - "-c", - program, - str(config.script), - str(ready), - *arguments, - "--project", - str(config.project), - ], - env=config.env, - stdin=subprocess.DEVNULL, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - text=True, - ) - deadline = time.monotonic() + 20 - while not ready.exists(): - assert process.poll() is None, process.communicate(timeout=20) - assert time.monotonic() < deadline, "Preference helper did not start" - time.sleep(0.01) - with pytest.raises(subprocess.TimeoutExpired): - process.communicate(timeout=0.5) - assert config.config.read_bytes() == before - assert config.agent().read_bytes() == agent_before - assert not config.agent(role="reviewer").exists() - finally: - release_file_lock(stream) - stdout, stderr = process.communicate(timeout=20) - assert process.returncode == 0, stderr - result = json.loads(stdout) - if action == "show": - assert result["worker"]["model"] == "sonnet" - assert config.config.read_bytes() == before - elif action == "set": - preferences = json.loads(config.config.read_text())["claude"] - assert set(preferences) == {"worker", "reviewer"} - assert preferences["worker"]["effort"] == "xhigh" - assert config.agent(role="reviewer").is_file() - assert config.agent().read_bytes() == agent_before - else: - assert not config.config.exists() - assert not config.agent().exists() - finally: - if process is not None and process.poll() is None: - process.kill() - process.communicate(timeout=20) - - -def test_show_reads_existing_lock_on_read_only_mount(config, configure_module, monkeypatch, capsys): - config.cli( - "set", - "--harness", - "codex", - "--role", - "worker", - "--model", - "gpt-6-luna", - "--effort", - "max", - user=True, - ) - lock = config.user_config.with_name(config.user_config.name + ".lock") - assert lock.is_file() - for name in ("XDG_CONFIG_HOME", "CLAUDE_CONFIG_DIR", "CODEX_HOME"): - monkeypatch.setenv(name, config.env[name]) - original_open = Path.open - - # Reject writes to the lock file to represent a read-only configuration mount. - def open_without_lock_writes(path, mode="r", *args, **kwargs): - if path == lock and "a" in mode: - raise OSError(errno.EROFS, "Read-only file system", str(path)) - return original_open(path, mode, *args, **kwargs) - - monkeypatch.setattr(Path, "open", open_without_lock_writes) - monkeypatch.setattr( - sys, - "argv", - [str(config.script), "show", "--harness", "codex", "--project", str(config.project)], - ) - configure_module.main() - assert json.loads(capsys.readouterr().out)["worker"] == { - "model": "gpt-6-luna", - "reasoning_effort": "max", - "allow_inherited_fallback": False, - } - - -def test_legacy_lock_directory_has_actionable_error(config): - lock = config.config.with_name(config.config.name + ".lock") - lock.mkdir() - assert "Legacy lock directory" in config.set_model(ok=False) - assert not config.config.exists() - - -def test_failed_config_write_restores_prior_agent(config, configure_module, monkeypatch): - config.set_model(model="old-model") - before_agent, before_config = config.agent().read_bytes(), config.config.read_bytes() - previous = configure_module.load_config(config.config) - desired = {"claude": {"worker": {"model": "new-model"}}} - original_replace = os.replace - - def fail_config(source, destination): - if Path(destination) == config.config: - raise OSError("simulated full disk") - original_replace(source, destination) - - monkeypatch.setattr(os, "replace", fail_config) - with pytest.raises(OSError, match="full disk"): - configure_module.update_scope(config.project, previous, desired) - assert config.agent().read_bytes() == before_agent - assert config.config.read_bytes() == before_config - assert not config.journal.exists() - - -@pytest.mark.parametrize("action", ["create", "update", "unconfigure"]) -@pytest.mark.parametrize("boundary", ["agent", "config"]) -def test_interrupted_writes_are_recovered_and_locks_released(config, action, boundary): - if action != "create": - config.set_model(model="old-model") - before_config = config.config.read_bytes() if config.config.exists() else None - before_agent = config.agent().read_bytes() if config.agent().exists() else None - arguments = ( - ("unconfigure",) - if action == "unconfigure" - else ( - "set", - "--harness", - "claude", - "--role", - "worker", - "--model", - "new-model", - ) - ) - config.interrupt(config.agent() if boundary == "agent" else config.config, *arguments) - assert config.journal.exists() - resolved = config.cli("show", "--harness", "claude") - assert resolved["worker"]["model"] == ("sonnet" if action == "create" else "old-model") - assert (config.config.read_bytes() if config.config.exists() else None) == before_config - assert (config.agent().read_bytes() if config.agent().exists() else None) == before_agent - assert not config.journal.exists() - config.set_model(model="next-model") - - -def test_interrupted_recovery_can_be_retried(config): - config.set_model(model="old-model") - config.interrupt( - config.agent(), "set", "--harness", "claude", "--role", "worker", "--model", "new-model" - ) - config.interrupt(config.agent(), "show", "--harness", "claude") - assert config.journal.exists() - assert config.cli("show", "--harness", "claude")["worker"]["model"] == "old-model" - assert not config.journal.exists() - - -@pytest.mark.parametrize("edited_file", ["agent", "config"]) -def test_recovery_preserves_post_crash_edits(config, edited_file): - config.set_model(model="old-model") - config.interrupt( - config.agent(), "set", "--harness", "claude", "--role", "worker", "--model", "new-model" - ) - target = config.agent() if edited_file == "agent" else config.config - target.write_text("Post-crash user edit") - before_agent = config.agent().read_bytes() - before_config = config.config.read_bytes() - assert "Preserving edited file during recovery" in config.cli( - "show", "--harness", "claude", ok=False - ) - assert config.agent().read_bytes() == before_agent - assert config.config.read_bytes() == before_config - assert config.journal.exists() - - -def test_recovery_rejects_unmanaged_paths(config): - outside = config.root / "outside" - outside.write_text("Untouched") - config.journal.write_text( - json.dumps( - { - "config": {"before": None, "after": None}, - "../outside": {"before": None, "after": outside.read_bytes().hex()}, - } - ) - ) - assert "Invalid recovery journal" in config.cli("show", "--harness", "claude", ok=False) - assert outside.read_text() == "Untouched" diff --git a/tests/test_smart_router.py b/tests/test_smart_router.py index e9e687d2d..a5bcee6ed 100644 --- a/tests/test_smart_router.py +++ b/tests/test_smart_router.py @@ -34,7 +34,7 @@ def test_smart_routed_session_installs_skill(tmp_path, monkeypatch, agent): home = config_io.APP_DIR.parent assert home.joinpath(f".{agent}/skills/{SMART_ROUTER_SKILL}/SKILL.md").is_file() assert home.joinpath(f".{agent}/skills/{ORCHESTRATOR_SKILL}/SKILL.md").is_file() - assert home.joinpath(f".{agent}/skills/{ORCHESTRATOR_SKILL}/scripts/configure.py").is_file() + assert home.joinpath(f".{agent}/skills/{ORCHESTRATOR_SKILL}/agents/reviewer.md").is_file() assert not home.joinpath(f".agents/skills/{SMART_ROUTER_SKILL}").exists() assert session_path == Path(os.environ[session_env.SESSION_ENV_VAR]) assert Path(os.environ[session_env.SESSION_ENV_VAR]).is_file() @@ -44,6 +44,10 @@ def test_smart_routed_session_installs_skill(tmp_path, monkeypatch, agent): @pytest.mark.skipif(os.name == "nt", reason="Exercises the skill's POSIX shell commands") @pytest.mark.parametrize("agent", ["claude", "codex"]) def test_skill_toggles_with_launch_installation_despite_shadowed_path(tmp_path, monkeypatch, agent): + # Launch preparation is reached only after routing eligibility has passed. + monkeypatch.setenv(v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR, "1") + monkeypatch.delenv(v2.ENABLE_SMART_ROUTING_ENV_VAR, raising=False) + monkeypatch.delenv("ISAAC_LAUNCH_MODE", raising=False) # Keep the venv path (including spaces), not its resolved system Python symlink. installation = tmp_path / "launch installation" installation.symlink_to(sys.prefix, target_is_directory=True) @@ -58,6 +62,10 @@ def test_skill_toggles_with_launch_installation_despite_shadowed_path(tmp_path, session_path = v2._prepare_smart_router_session(agent) skill = config_io.APP_DIR.parent / f".{agent}/skills/{SMART_ROUTER_SKILL}/SKILL.md" commands = re.findall(r"`([^`\n]*--(?:enable|disable)-smart-routing)`", skill.read_text()) + orchestrate = config_io.APP_DIR.parent / f".{agent}/skills/{ORCHESTRATOR_SKILL}/SKILL.md" + check_command = re.search( + r'^"\$UCODE_SMART_ROUTER_PYTHON" -m [^\n]+ --check$', orchestrate.read_text(), re.M + ).group() for action in ("disable", "enable"): command = next(cmd for cmd in commands if f" {agent} --{action}-" in cmd) @@ -74,6 +82,21 @@ def test_skill_toggles_with_launch_installation_despite_shadowed_path(tmp_path, assert f"Smart Router is {'off' if action == 'disable' else 'on'}" in result.stdout expected = dict.fromkeys(v2.SMART_ROUTING_ENV_KEYS, "0") if action == "disable" else {} assert json.loads(session_path.read_text()) == expected + check = subprocess.run( + ["/bin/sh", "-c", check_command], + cwd=tmp_path, + env={**os.environ, "NO_COLOR": "1"}, + stdin=subprocess.DEVNULL, + capture_output=True, + text=True, + timeout=20, + ) + assert check.returncode == (1 if action == "disable" else 0), check.stderr + assert check.stdout == "" + if action == "disable": + assert "smart routing is off" in check.stderr + else: + assert check.stderr == "" def test_launcher_flags_control_routing_hook(tmp_path, monkeypatch): From d6131c3546e2a67f6c128a6348e0478c077a16f5 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 16:47:42 +0000 Subject: [PATCH 03/25] Preserve bundled orchestrator role defaults --- skills/orchestrate/agents/explorer.md | 2 +- skills/orchestrate/agents/researcher.md | 2 +- skills/orchestrate/agents/reviewer.md | 2 +- skills/orchestrate/agents/tester.md | 2 +- skills/orchestrate/agents/worker.md | 2 +- 5 files changed, 5 insertions(+), 5 deletions(-) diff --git a/skills/orchestrate/agents/explorer.md b/skills/orchestrate/agents/explorer.md index d030391a6..4d68fe44b 100644 --- a/skills/orchestrate/agents/explorer.md +++ b/skills/orchestrate/agents/explorer.md @@ -1,7 +1,7 @@ --- name: explorer description: Map code, callers, existing patterns, and tests for a bounded investigation without editing files. -model: inherit +model: sonnet --- Follow the supervisor's bounded task contract. Inspect source and callers; report diff --git a/skills/orchestrate/agents/researcher.md b/skills/orchestrate/agents/researcher.md index b157d616f..5c4783e84 100644 --- a/skills/orchestrate/agents/researcher.md +++ b/skills/orchestrate/agents/researcher.md @@ -1,7 +1,7 @@ --- name: researcher description: Verify external documentation and API facts against primary sources without editing files. -model: inherit +model: sonnet --- Follow the supervisor's bounded task. Cite source links; distinguish observations diff --git a/skills/orchestrate/agents/reviewer.md b/skills/orchestrate/agents/reviewer.md index bb224bac4..14d9e3eaf 100644 --- a/skills/orchestrate/agents/reviewer.md +++ b/skills/orchestrate/agents/reviewer.md @@ -1,7 +1,7 @@ --- name: reviewer description: Independently inspect the resulting implementation for correctness, regressions, security, and missing tests. -model: inherit +model: sonnet --- Follow the supervisor's bounded task contract. Read the changed code and relevant diff --git a/skills/orchestrate/agents/tester.md b/skills/orchestrate/agents/tester.md index 6a0b294c2..b9595e391 100644 --- a/skills/orchestrate/agents/tester.md +++ b/skills/orchestrate/agents/tester.md @@ -1,7 +1,7 @@ --- name: tester description: Independently execute acceptance checks and diagnose failures; edit tests only when assigned. -model: inherit +model: sonnet --- Follow the supervisor's bounded task contract. Run the requested checks and report diff --git a/skills/orchestrate/agents/worker.md b/skills/orchestrate/agents/worker.md index 7d95005a6..3cb241176 100644 --- a/skills/orchestrate/agents/worker.md +++ b/skills/orchestrate/agents/worker.md @@ -1,7 +1,7 @@ --- name: worker description: Implement one bounded change in files explicitly assigned by the supervisor. -model: inherit +model: sonnet --- Follow the supervisor's bounded task contract and file ownership. Preserve other From c153167fd79e10258379c0e32905dab19c26a490 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 16:59:35 +0000 Subject: [PATCH 04/25] Let Codex merge native hook sources --- AGENTS.md | 2 +- README.md | 6 +- src/ucode/codex_config.py | 103 -------------------- src/ucode/smart_routing/v2.py | 33 ++----- tests/README.md | 8 +- tests/integration/README.md | 7 +- tests/test_codex_config.py | 136 --------------------------- tests/test_codex_smart_routing_v2.py | 94 +++++++++--------- tests/test_orchestrator.py | 13 ++- 9 files changed, 71 insertions(+), 331 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 93ad62b4d..e219fb94e 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -88,7 +88,7 @@ Fields live in `~/.codex/ucode.config.toml` and `/etc/codex/managed_config.toml` | `http_headers` | Merge | Merge | In `[model_providers.Databricks]`; merge `ug`'s routing headers by name, admin headers added under managed config | | `model_catalog_json` | Create/replace | Create/replace | In `~/.codex/config.toml`; `ug`'s own catalog reference, for a static model list | | `mcp_servers` | Ignore | Merge | Managed file; add/update the config's MCP server entries, other entries left alone | -| Smart-routing and orchestrator hooks | Merge | Merge | Launch-only hooks preserve native effective handlers, including trusted project hooks; only `PreToolUse`, `UserPromptSubmit`, and `SessionStart` are overridden; `features.hooks` is enabled for a smart-routed launch | +| Smart-routing and orchestrator hooks | Merge | Merge | Launch-only `PreToolUse`, `UserPromptSubmit`, and `SessionStart` handlers; Codex combines them with its native hook sources and applies project trust; `features.hooks` is enabled for a smart-routed launch | | `plugins."model-orchestrator@…".enabled` | Merge | Merge | Launch-only `false` overrides for legacy registrations, including Isaac-synced and project plugins, even when smart routing is off; saved settings and other plugins left alone | diff --git a/README.md b/README.md index 14f61bfbb..40d8d514a 100644 --- a/README.md +++ b/README.md @@ -265,9 +265,9 @@ are unchanged. UG automatically suppresses installed standalone `model-orchestrator` plugins, including Isaac-synced Codex registrations, for each Claude or Codex launch. This also applies when smart routing is off, so the old hooks cannot activate -orchestration independently. Codex project registrations are included. Routed Codex -launches use its native configuration resolver to preserve trusted project hooks; -untrusted project hooks stay disabled. Saved plugin settings and unrelated plugins +orchestration independently. Codex project registrations are included. UG supplies +only its own hooks; Codex combines them with existing hooks and applies project +trust. Saved plugin settings and unrelated plugins and hooks remain intact. Smart routing selects subagent models; separate role-model preferences are ignored and their files are left untouched. See the bundled [orchestrator documentation](skills/orchestrate/README.md) for cutover details. diff --git a/src/ucode/codex_config.py b/src/ucode/codex_config.py index f9af79907..c144d6af3 100644 --- a/src/ucode/codex_config.py +++ b/src/ucode/codex_config.py @@ -2,14 +2,8 @@ from __future__ import annotations -import json import os -import queue -import subprocess -import threading -import time from collections.abc import Mapping -from contextlib import suppress from enum import StrEnum from pathlib import Path @@ -18,12 +12,10 @@ from ucode.config_io import read_json_safe, read_toml_safe from ucode.managed_files import OS, current_os -from ucode.os_compatibility import subprocess_cross_os from ucode.ui import print_warning CODEX_PROFILE_NAME = "ucode" DEFAULT_CODEX_CONFIG_PATH = Path.home() / ".codex" / f"{CODEX_PROFILE_NAME}.config.toml" -CONFIG_READ_TIMEOUT_SECONDS = 15 def codex_working_directory(tool_args: list[str]) -> Path: @@ -44,98 +36,6 @@ def codex_working_directory(tool_args: list[str]) -> Path: return directory.resolve() -def codex_cli_config_args(tool_args: list[str]) -> list[str]: - """Keep caller configuration overrides, including project trust, in the native lookup.""" - config_args = [] - args = iter(tool_args) - options = {"-c", "--config", "--enable", "--disable"} - for arg in args: - if arg == "--": - break - if arg in options: - value = next(args, None) - if value is not None: - config_args.extend([arg, value]) - elif arg.partition("=")[0] in options or arg.startswith("-c"): - config_args.append(arg) - return config_args - - -def read_effective_codex_config(binary: str, *, cwd: Path, config_args: list[str]) -> dict: - """Let Codex apply its configuration precedence and project trust rules.""" - error = ( - "Could not read Codex configuration for smart routing. Check your Codex configuration " - "or launch with --disable-smart-routing." - ) - try: - process = subprocess_cross_os.popen( - [binary, "app-server", *config_args, "--listen", "stdio://"], - cwd=cwd, - stdin=subprocess.PIPE, - stdout=subprocess.PIPE, - stderr=subprocess.DEVNULL, - text=True, - ) - except OSError as exc: - raise RuntimeError(error) from exc - stdin, stdout = process.stdin, process.stdout - assert stdin is not None and stdout is not None - messages: queue.Queue = queue.Queue() - - def read_messages() -> None: - try: - for line in stdout: - messages.put(json.loads(line)) - except (OSError, ValueError): - pass - finally: - messages.put(None) - - reader = threading.Thread(target=read_messages, daemon=True) - reader.start() - deadline = time.monotonic() + CONFIG_READ_TIMEOUT_SECONDS - - def request(request_id: int, method: str, params: dict) -> dict: - stdin.write(json.dumps({"id": request_id, "method": method, "params": params}) + "\n") - stdin.flush() - while time.monotonic() < deadline: - message = messages.get(timeout=max(0, deadline - time.monotonic())) - if not isinstance(message, dict): - raise RuntimeError(error) - if message.get("id") == request_id: - result = message.get("result") - if not isinstance(result, dict) or "error" in message: - raise RuntimeError(error) - return result - raise RuntimeError(error) - - try: - request(1, "initialize", {"clientInfo": {"name": "unity-gateway", "version": "1"}}) - stdin.write('{"method":"initialized","params":{}}\n') - stdin.flush() - result = request(2, "config/read", {"cwd": str(cwd), "includeLayers": False}) - config = result.get("config") - if not isinstance(config, dict): - raise RuntimeError(error) - return config - except (OSError, queue.Empty) as exc: - raise RuntimeError(error) from exc - finally: - with suppress(OSError): - stdin.close() - try: - process.wait(timeout=5) - except subprocess.TimeoutExpired: - process.terminate() - try: - process.wait(timeout=5) - except subprocess.TimeoutExpired: - process.kill() - process.wait(timeout=5) - reader.join(timeout=5) - stdout.close() - - class ModelVisibility(StrEnum): """A model's visibility in Codex's picker/APIs (mirrors Codex's ModelVisibility).""" @@ -229,9 +129,6 @@ def _toml_item(value: object) -> Item: if isinstance(value, Mapping): inline = tomlkit.inline_table() for key, child in value.items(): - # Native config/read includes null optional fields; TOML represents them by omission. - if child is None: - continue inline[str(key)] = _toml_item(child) return inline if isinstance(value, list): diff --git a/src/ucode/smart_routing/v2.py b/src/ucode/smart_routing/v2.py index e8eaf299c..ba5515fa6 100644 --- a/src/ucode/smart_routing/v2.py +++ b/src/ucode/smart_routing/v2.py @@ -10,19 +10,16 @@ import time import urllib.request from collections.abc import Callable, MutableMapping -from copy import deepcopy from pathlib import Path from tempfile import TemporaryDirectory from typing import NoReturn, TextIO from ucode import config_io from ucode.codex_config import ( - codex_cli_config_args, codex_config_args, codex_working_directory, custom_catalog_models, custom_catalog_path, - read_effective_codex_config, ) from ucode.config_io import ( APP_DIR, @@ -609,23 +606,13 @@ def _cached_routing_models(state: dict) -> list[str]: return routing_models(state) -def _v2_hooks(state: dict, available_models: list[str], config: dict) -> dict: - configured_hooks = config.get("hooks") - events = ("PreToolUse", "UserPromptSubmit", "SessionStart") - # Only override events UG changes. Codex retains ownership of every other event. +def _v2_hooks(state: dict, available_models: list[str]) -> dict: + # Codex combines hook sources itself; copying user hooks here would register them twice. doc = { "hooks": { - event: deepcopy(configured_hooks[event]) - for event in events - if isinstance(configured_hooks, dict) and event in configured_hooks + "PreToolUse": merge_pre_tool_use_hooks([], state, available_models=available_models), } } - existing = doc["hooks"].get("PreToolUse") - doc["hooks"]["PreToolUse"] = merge_pre_tool_use_hooks( - existing if isinstance(existing, list) else [], - state, - available_models=available_models, - ) orchestrator.sync_hooks(doc, agent="codex") return doc["hooks"] @@ -668,15 +655,11 @@ def launch_codex( catalog_path = custom_catalog_path() if catalog_path is not None: overlay["model_catalog_json"] = str(catalog_path) - cwd = codex_working_directory(tool_args) - config = read_effective_codex_config( - binary, - cwd=cwd, - config_args=[*codex_config_args(overlay), *codex_cli_config_args(tool_args)], - ) - overlay["hooks"] = _v2_hooks(state, available_models, config) + overlay["hooks"] = _v2_hooks(state, available_models) overlay["features.hooks"] = True - legacy_plugin_config = orchestrator.legacy_codex_plugin_config(cwd=cwd) + legacy_plugin_config = orchestrator.legacy_codex_plugin_config( + cwd=codex_working_directory(tool_args) + ) overlay.update(legacy_plugin_config) _prepare_smart_router_session("codex") # Codex constructs tool subprocess environments through its shell policy. @@ -687,7 +670,7 @@ def launch_codex( overlay[f"shell_environment_policy.set.{key}"] = os.environ[key] config_args = codex_config_args(overlay) if not first_prompt_routing_enabled(): - # Subagent-only routing needs no persistent app-server or interposer: + # Subagent-only routing needs neither the app-server nor the interposer: # the hooks ride in the CLI config, so launch the TUI directly. exec_or_spawn([binary, *config_args, *tool_args]) app_port = _free_port() diff --git a/tests/README.md b/tests/README.md index 98cbf986b..07378a861 100644 --- a/tests/README.md +++ b/tests/README.md @@ -144,9 +144,11 @@ and the Claude/Codex launcher tests check native per-launch overrides that disab legacy marketplace registrations with routing on or off, including Codex's app-server and remote TUI. They check config discovery, unrelated-plugin and hook preservation, and unchanged saved settings, including nested project registrations -and `--cd`. `test_codex_config.py` exercises the native configuration protocol, -timeout cleanup, and optional hook fields; launch tests preserve its resolved -handlers and leave unrelated hook events to Codex. These are component assertions. +and `--cd`. `test_codex_config.py` covers launch-directory resolution and argument +serialization. Codex launch tests check that UG supplies only its own hooks, +preserves caller arguments and saved configuration, and starts no helper process +for subagent-only routing. Native hook merging and project trust belong to Codex; +these component assertions do not exercise its hook loader. The toggle integration journeys require both bundled skills, verify the saved session controls and native tool-result confirmation after each toggle, and explicitly request their children, including while routing is off. Live automatic diff --git a/tests/integration/README.md b/tests/integration/README.md index b8766477f..3c8381afe 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -295,9 +295,10 @@ session toggles. Separate preference editing, locking, and recovery are not part of the workflow. `../test_orchestrator_legacy_plugins.py` and launcher component tests check per-launch suppression of installed legacy plugins, including non-routed launches, nested project config and `--cd`, and preservation -of saved settings and unrelated plugins/hooks. Native config protocol and timeout -handling have component coverage in `../test_codex_config.py`; native project trust -and hook execution are not exercised by this integration suite. Live automatic +of saved settings and unrelated plugins/hooks. Codex launcher component tests +check that UG supplies only its own hooks, preserves caller arguments, and starts +no helper process for subagent-only routing. Codex's native hook merging, project +trust, and execution of pre-existing hooks are not exercised by this integration suite. Live automatic delegation and legacy-hook execution remain unverified by this suite. diff --git a/tests/test_codex_config.py b/tests/test_codex_config.py index 219e626f4..918d1fa09 100644 --- a/tests/test_codex_config.py +++ b/tests/test_codex_config.py @@ -1,10 +1,5 @@ from __future__ import annotations -import json -import subprocess -import sys -import textwrap - import pytest import tomlkit @@ -68,32 +63,6 @@ def test_renders_nested_tables_from_parsed_profile(self): assert 'auth = {command = "ucode", args = ["codex-token"]}' in provider_override assert 'tui={model_availability_nux = {"gpt-5.6-sol" = 1}}' in args - def test_renders_native_hook_defaults_as_optional_fields(self): - config = { - "hooks": { - "UserPromptSubmit": [ - { - "matcher": None, - "hooks": [ - {"type": "command", "command": "policy", "command_windows": None} - ], - } - ] - } - } - - args = codex_config_args(config) - - assert tomlkit.parse(args[1]) == { - "hooks": { - "UserPromptSubmit": [ - { - "hooks": [{"type": "command", "command": "policy"}], - } - ] - } - } - @pytest.mark.parametrize( "args", @@ -109,108 +78,3 @@ def test_renders_native_hook_defaults_as_optional_fields(self): def test_codex_working_directory(tmp_path, monkeypatch, args): monkeypatch.chdir(tmp_path) assert codex_config.codex_working_directory(args) == tmp_path / "project" - - -def test_native_lookup_preserves_caller_config_overrides(): - overrides = [ - "-c", - 'projects={"/project"={trust_level="untrusted"}}', - "--config", - 'model="example"', - "--config=features.hooks=true", - "-cfeatures.search=false", - "--disable", - "remote_control", - ] - args = ["--cd", "/project", *overrides, "--", "--config=prompt-text"] - - assert codex_config.codex_cli_config_args(args) == overrides - assert args[-1] == "--config=prompt-text" - - -class TestReadEffectiveCodexConfig: - def _server(self, monkeypatch, script): - processes = [] - calls = [] - - def start(argv, **kwargs): - # Substitute only the native protocol peer; exercise real pipes and cleanup. - calls.append((argv, kwargs)) - process = subprocess.Popen([sys.executable, "-c", textwrap.dedent(script)], **kwargs) - processes.append(process) - return process - - monkeypatch.setattr(codex_config.subprocess_cross_os, "popen", start) - return processes, calls - - def test_reads_native_result_after_handshake(self, tmp_path, monkeypatch, capsys): - config = {"hooks": {"UserPromptSubmit": [{"hooks": [{"command": "project-policy"}]}]}} - script = """ - import json, sys - initial = json.loads(sys.stdin.readline()) - assert initial['method'] == 'initialize' - print(json.dumps({'id': initial['id'], 'result': {}}), flush=True) - assert json.loads(sys.stdin.readline())['method'] == 'initialized' - request = json.loads(sys.stdin.readline()) - assert request['method'] == 'config/read' - assert request['params'] == {'cwd': CWD, 'includeLayers': False} - print(json.dumps({'method': 'notification'}), flush=True) - print(json.dumps({'id': request['id'], 'result': {'config': CONFIG}}), flush=True) - assert sys.stdin.read() == '' - """.replace("CWD", repr(str(tmp_path))).replace("CONFIG", repr(config)) - processes, calls = self._server(monkeypatch, script) - - assert ( - codex_config.read_effective_codex_config( - "/selected/codex", cwd=tmp_path, config_args=["--config", 'model="example"'] - ) - == config - ) - - assert calls[0][0] == [ - "/selected/codex", - "app-server", - "--config", - 'model="example"', - "--listen", - "stdio://", - ] - assert calls[0][1]["cwd"] == tmp_path - assert processes[0].returncode == 0 - assert processes[0].stdin.closed and processes[0].stdout.closed - assert capsys.readouterr().out == "" - - @pytest.mark.parametrize( - "response", - [ - "not json", - json.dumps({"id": 1, "error": {"message": "invalid config"}}), - json.dumps({"id": 1, "result": None}), - ], - ) - def test_protocol_errors_do_not_fall_back_to_incomplete_hooks( - self, tmp_path, monkeypatch, response - ): - processes, _ = self._server( - monkeypatch, - f""" - import sys - sys.stdin.readline() - print({response!r}, flush=True) - """, - ) - - with pytest.raises(RuntimeError, match="--disable-smart-routing"): - codex_config.read_effective_codex_config("codex", cwd=tmp_path, config_args=[]) - - assert processes[0].poll() is not None - - def test_unresponsive_server_is_reaped(self, tmp_path, monkeypatch): - monkeypatch.setattr(codex_config, "CONFIG_READ_TIMEOUT_SECONDS", 0.05) - processes, _ = self._server(monkeypatch, "import time; time.sleep(60)") - - with pytest.raises(RuntimeError, match="Could not read Codex configuration"): - codex_config.read_effective_codex_config("codex", cwd=tmp_path, config_args=[]) - - assert processes[0].poll() is not None - assert processes[0].stdin.closed and processes[0].stdout.closed diff --git a/tests/test_codex_smart_routing_v2.py b/tests/test_codex_smart_routing_v2.py index d999d3638..b3496d86b 100644 --- a/tests/test_codex_smart_routing_v2.py +++ b/tests/test_codex_smart_routing_v2.py @@ -3,7 +3,6 @@ import json import os import tomllib -from copy import deepcopy from types import SimpleNamespace import pytest @@ -15,14 +14,6 @@ WS = "https://example.databricks.com" -@pytest.fixture(autouse=True) -def native_config(monkeypatch): - # Launch tests isolate the native Codex process; its protocol is tested separately. - config = {} - monkeypatch.setattr(v2, "read_effective_codex_config", lambda *args, **kwargs: config) - return config - - def test_smart_routing_switch_message_is_boxed(): message = v2.format_routing_notice("model-x", "Because X.") @@ -474,7 +465,7 @@ def fake_exec(argv): assert os.environ[v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR] == "1" assert os.environ[v2.OAUTH_TOKEN_ENV_VAR] == "token" - def test_v2_pre_tool_hook_preserves_user_hooks(self, tmp_path, monkeypatch): + def test_v2_pre_tool_hook_leaves_saved_hooks_to_codex(self, tmp_path, monkeypatch): codex_home = tmp_path / ".codex" codex_home.mkdir() (codex_home / "config.toml").write_text( @@ -487,78 +478,82 @@ def test_v2_pre_tool_hook_preserves_user_hooks(self, tmp_path, monkeypatch): ) monkeypatch.setenv("CODEX_HOME", str(codex_home)) + before = (codex_home / "config.toml").read_bytes() configured = v2._v2_hooks( {"workspace": WS, "profile": "myprof"}, ["system.ai.gpt-5-6-sol"], - tomllib.loads((codex_home / "config.toml").read_text()), )["PreToolUse"] - assert configured[0]["hooks"][0]["command"] == "user-policy" - assert configured[1]["matcher"] == "Agent|.*spawn_agent$" - assert "--model system.ai.gpt-5-6-sol" in configured[1]["hooks"][0]["command"] + assert len(configured) == 1 + assert configured[0]["matcher"] == "Agent|.*spawn_agent$" + assert "--model system.ai.gpt-5-6-sol" in configured[0]["hooks"][0]["command"] + assert (codex_home / "config.toml").read_bytes() == before - def test_subagent_launch_composes_only_native_resolved_events(self, tmp_path, monkeypatch): + def test_subagent_launch_leaves_native_hooks_to_codex(self, tmp_path, monkeypatch): monkeypatch.setenv(v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR, "1") monkeypatch.setattr(v2, "get_databricks_token", lambda *args: "token") + monkeypatch.setattr( + v2.subprocess_cross_os, + "popen", + lambda *args, **kwargs: pytest.fail("subagent-only launch needs no helper process"), + ) + user_home = tmp_path / "user" + user_home.mkdir() + monkeypatch.setenv("CODEX_HOME", str(user_home)) project = tmp_path / "project" (project / ".codex").mkdir(parents=True) + user_config = user_home / "config.toml" project_config = project / ".codex/config.toml" - project_config.write_text('[plugins."model-orchestrator@project"]\nenabled = true\n') - before = project_config.read_bytes() - effective = { - "hooks": { - event: [{"hooks": [{"type": "command", "command": f"native-{event}"}]}] - for event in ("PreToolUse", "UserPromptSubmit", "SessionStart", "Stop") - } - } - saved = deepcopy(effective) - reads, launches = [], [] + for source, path in (("user", user_config), ("project", project_config)): + path.write_text( + "".join( + f"[[hooks.{event}]]\n[[hooks.{event}.hooks]]\n" + f'type = "command"\ncommand = "{source}-{event}"\n' + for event in ("PreToolUse", "UserPromptSubmit", "SessionStart", "Stop") + ) + + '[plugins."model-orchestrator@project"]\nenabled = true\n' + ) + before = {path: path.read_bytes() for path in (user_config, project_config)} + launches = [] caller_config = ["--config", 'projects={"/other"={trust_level="untrusted"}}'] - - def native_read(binary, **kwargs): - reads.append((binary, kwargs)) - return effective + tool_args = [*caller_config, "--cd", str(project)] def launch(argv): launches.append(argv) raise SystemExit(0) - monkeypatch.setattr(v2, "read_effective_codex_config", native_read) monkeypatch.setattr(v2, "exec_or_spawn", launch) with pytest.raises(SystemExit): v2.launch_codex( {"workspace": WS, "codex_models": ["gpt-5.6-sol"]}, - [*caller_config, "--cd", str(project)], + tool_args, binary="/selected/codex", start_model="gpt-5.6-sol", render_overlay=lambda *args, **kwargs: {"model": "gpt-5.6-sol"}, ) - assert reads == [ - ( - "/selected/codex", - { - "cwd": project, - "config_args": ["--config", 'model="gpt-5.6-sol"', *caller_config], - }, - ) - ] (argv,) = launches - for event in ("PreToolUse", "UserPromptSubmit", "SessionStart"): + expected_handlers = { + "PreToolUse": "codex-router-hook route-subagent", + "UserPromptSubmit": "ucode.smart_routing.orchestrator", + "SessionStart": "ucode.smart_routing.orchestrator", + } + assert {arg.split("=", 1)[0] for arg in argv if arg.startswith("hooks.")} == { + f"hooks.{event}" for event in expected_handlers + } + for event, handler in expected_handlers.items(): value = next(arg for arg in argv if arg.startswith(f"hooks.{event}=")) - groups = tomllib.loads(value)["hooks"][event] - assert groups[0] == saved["hooks"][event][0] - assert len(groups) == 2 - assert not any(arg.startswith("hooks.Stop=") for arg in argv) + (group,) = tomllib.loads(value)["hooks"][event] + (hook,) = group["hooks"] + assert handler in hook["command"] plugin_arg = next(arg for arg in argv if arg.startswith("plugins=")) assert tomllib.loads(plugin_arg) == { "plugins": {"model-orchestrator@project": {"enabled": False}} } - assert argv[-2:] == ["--cd", str(project)] - assert effective == saved - assert project_config.read_bytes() == before + assert argv[-len(tool_args) :] == tool_args + assert all(path.read_bytes() == content for path, content in before.items()) - def test_v2_pre_tool_hook_replaces_existing_ucode_hook(self, tmp_path, monkeypatch): + def test_v2_pre_tool_hook_uses_current_model(self, tmp_path, monkeypatch): monkeypatch.setattr("ucode.databricks.ug_binary", lambda: "/bin/ug") codex_home = tmp_path / ".codex" codex_home.mkdir() @@ -575,7 +570,6 @@ def test_v2_pre_tool_hook_replaces_existing_ucode_hook(self, tmp_path, monkeypat configured = v2._v2_hooks( {"workspace": WS, "profile": "myprof"}, ["system.ai.gpt-5-6-sol"], - tomllib.loads((codex_home / "config.toml").read_text()), )["PreToolUse"] routing_commands = [ diff --git a/tests/test_orchestrator.py b/tests/test_orchestrator.py index 96b9071bc..928ac5804 100644 --- a/tests/test_orchestrator.py +++ b/tests/test_orchestrator.py @@ -215,7 +215,7 @@ def test_hooks_preserve_user_handlers_and_replace_only_ug_handlers(monkeypatch): assert doc["hooks"]["SessionStart"][1]["matcher"] == "compact" -def test_codex_launch_merges_prompt_and_compaction_hooks(tmp_path, monkeypatch): +def test_codex_launch_adds_only_its_prompt_and_compaction_hooks(tmp_path, monkeypatch): config = tmp_path / "config.toml" config.write_text( '[[hooks.UserPromptSubmit]]\n[[hooks.UserPromptSubmit.hooks]]\ncommand = "user-prompt"\n' @@ -224,12 +224,11 @@ def test_codex_launch_merges_prompt_and_compaction_hooks(tmp_path, monkeypatch): ) before = config.read_bytes() monkeypatch.setenv("CODEX_HOME", str(tmp_path)) - hooks = v2._v2_hooks( - {"workspace": "https://example.com"}, ["gpt-6-sol"], v2.config_io.read_toml_safe(config) - ) - assert hooks["UserPromptSubmit"][0]["hooks"][0]["command"] == "user-prompt" - assert hooks["SessionStart"][0]["hooks"][0]["command"] == "user-compact" - assert orchestrator.HOOK_MODULE in hooks["UserPromptSubmit"][1]["hooks"][0]["command"] + hooks = v2._v2_hooks({"workspace": "https://example.com"}, ["gpt-6-sol"]) + for event in ("UserPromptSubmit", "SessionStart"): + (group,) = hooks[event] + (hook,) = group["hooks"] + assert orchestrator.HOOK_MODULE in hook["command"] assert config.read_bytes() == before From 4ee23bbe8d8623f634dda6daa21bc58c455bc28c Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 17:03:39 +0000 Subject: [PATCH 05/25] Handle orchestrator check confirmation in routing journey --- tests/README.md | 5 ++- tests/integration/README.md | 5 +++ .../test_ug_smart_routing_hooks.py | 29 +++++++++++++-- tests/integration/utils/evidence.py | 21 +++++++++++ tests/test_integration_evidence.py | 35 +++++++++++++++++++ 5 files changed, 92 insertions(+), 3 deletions(-) diff --git a/tests/README.md b/tests/README.md index 07378a861..e9f9839c4 100644 --- a/tests/README.md +++ b/tests/README.md @@ -151,7 +151,10 @@ for subagent-only routing. Native hook merging and project trust belong to Codex these component assertions do not exercise its hook loader. The toggle integration journeys require both bundled skills, verify the saved session controls and native tool-result confirmation after each toggle, and -explicitly request their children, including while routing is off. Live automatic +explicitly request their children, including while routing is off. Claude's journey +answers the visible permission prompt for the exact read-only orchestrator check, +verified against the native pending command. Evidence tests reject added shell +commands and other permission selections. Live automatic delegation and legacy-hook execution are not covered by these tests. `test_integration_evidence.py` checks native tool-result extraction for both agents, including collapsed-output records, and excludes user echoes and assistant claims. diff --git a/tests/integration/README.md b/tests/integration/README.md index 3c8381afe..e2f27ae51 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -302,6 +302,11 @@ trust, and execution of pre-existing hooks are not exercised by this integration delegation and legacy-hook execution remain unverified by this suite. +The Claude on/off/on journey handles the visible permission prompt for the exact +read-only orchestrator check. It compares the prompt with the native pending +command before selecting the one-time Yes option, then still requires the routed +banner, completed child, and correlated routing decision. + The portable `../test_claude_windows_smart_routing.py` checks the Windows subagent-only fallback without Unix imports. Native Windows TUI and hook execution remain outside this integration suite. diff --git a/tests/integration/test_ug_smart_routing_hooks.py b/tests/integration/test_ug_smart_routing_hooks.py index 7fc8ecb8e..f5e151bca 100644 --- a/tests/integration/test_ug_smart_routing_hooks.py +++ b/tests/integration/test_ug_smart_routing_hooks.py @@ -19,6 +19,7 @@ assert_subagent_routed, assistant_answers, is_child_session, + is_orchestrator_check_permission, read_jsonl, tool_outputs, ) @@ -90,8 +91,31 @@ def _run_calculation(tui, session, agent: str, expression: str, expected: str, * tui.submit(task.prompt) if routed: + permission_in_progress = False + + def routed_banner_visible(screen): + nonlocal permission_in_progress + if "Do you want to proceed?" in screen: + if permission_in_progress: + return False + roots = [ + records + for path, records in agent_sessions(session, agent).items() + if not is_child_session(agent, path, records) + ] + assert ( + agent == "claude" + and len(roots) == 1 + and is_orchestrator_check_permission(screen, roots[0]) + ), "Unrecognized permission before subagent routing:\n" + screen + tui.send("\r", "allow the read-only orchestrator routing-state check") + permission_in_progress = True + return False + permission_in_progress = False + return _routing_banner_for_task(screen, task.marker) + tui.wait_for( - lambda screen: _routing_banner_for_task(screen, task.marker), + routed_banner_visible, f"the Smart Router subagent banner for {task.marker}", timeout=120, ) @@ -289,7 +313,8 @@ def test_smart_router_skill_toggles_claude_subagent_routing(live_session, worksp routing controls, even with collapsed terminal output; all three uniquely tagged calculations complete in native child sessions; only the first and third show the subagent-routing banner and produce live gateway decisions correlated with those children. - No first-prompt routing wrapper starts. + A visible permission prompt for the exact read-only orchestrator check is accepted; + other commands are rejected. No first-prompt routing wrapper starts. """ session = live_session session.env["TMPDIR"] = str(tmp_path) diff --git a/tests/integration/utils/evidence.py b/tests/integration/utils/evidence.py index d3f77a893..3cd988aa7 100644 --- a/tests/integration/utils/evidence.py +++ b/tests/integration/utils/evidence.py @@ -88,6 +88,27 @@ def is_child_session(agent: str, path: str, records: list[dict]) -> bool: return _AGENT_HELPERS.get(agent, codex).is_child_session(path, records) +def is_orchestrator_check_permission(screen: str, records: list[dict]) -> bool: + """Match the visible Claude prompt against its actual pending command.""" + command = '"$UCODE_SMART_ROUTER_PYTHON" -m ucode.smart_routing.orchestrator --check' + if not ( + "Do you want to proceed?" in screen + and re.search(r"(?m)^\s*Bash command\s*$", screen) + and re.search(r"(?m)^\s*[›❯>]\s*1\.\s*Yes\s*$", screen) + and re.search(rf"(?m)^[ \t]*{re.escape(command)}[ \t]*$", screen) + ): + return False + for record in reversed(records): + if record.get("type") != "assistant": + continue + for part in reversed(record.get("message", {}).get("content", [])): + if isinstance(part, dict) and part.get("type") == "tool_use": + return ( + part.get("name") == "Bash" and part.get("input", {}).get("command") == command + ) + return False + + def completed_task_models(session, agent: str, answer_value: str) -> set[str]: """Read parent task model evidence, not proof of the gateway's destination.""" assert agent in _AGENT_HELPERS, f"Unsupported evidence agent: {agent}" diff --git a/tests/test_integration_evidence.py b/tests/test_integration_evidence.py index cf04efc76..ec793a4e6 100644 --- a/tests/test_integration_evidence.py +++ b/tests/test_integration_evidence.py @@ -85,6 +85,41 @@ def test_transient_retries_and_running_tasks_are_not_terminal_errors(screen): assert_no_terminal_api_error(screen) +@pytest.mark.parametrize( + "command_suffix,tool_name,selected,accepted", + [ + ("", "Bash", "1. Yes", True), + ("; echo unrelated", "Bash", "1. Yes", False), + ("\necho unrelated", "Bash", "1. Yes", False), + ("", "Read", "1. Yes", False), + ("", "Bash", "2. Yes, and switch to auto mode", False), + ], +) +def test_orchestrator_permission_requires_exact_pending_command( + command_suffix, tool_name, selected, accepted +): + command = '"$UCODE_SMART_ROUTER_PYTHON" -m ucode.smart_routing.orchestrator --check' + screen = f"Bash command\n {command}\nDo you want to proceed?\n❯ {selected}\n" + records = [ + { + "type": "assistant", + "message": { + "content": [ + {"type": "tool_use", "name": "Bash", "input": {"command": command}}, + { + "type": "tool_use", + "name": tool_name, + "input": {"command": command + command_suffix}, + }, + ] + }, + } + ] + + assert evidence.is_orchestrator_check_permission(screen, records) is accepted + assert not evidence.is_orchestrator_check_permission(screen, []) + + @pytest.mark.parametrize("agent", ["claude", "codex"]) def test_tagged_calculation_requires_the_native_child_answer(tmp_path, agent): session = _Session(tmp_path) From 59a4fd168ddd1f8d5452aee720956200c06eb78a Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 17:12:45 +0000 Subject: [PATCH 06/25] Wait for the orchestrator command transcript before confirming --- tests/README.md | 3 ++- tests/integration/README.md | 3 ++- tests/integration/test_ug_smart_routing_hooks.py | 6 ++++-- 3 files changed, 8 insertions(+), 4 deletions(-) diff --git a/tests/README.md b/tests/README.md index e9f9839c4..0d017dec3 100644 --- a/tests/README.md +++ b/tests/README.md @@ -153,7 +153,8 @@ The toggle integration journeys require both bundled skills, verify the saved session controls and native tool-result confirmation after each toggle, and explicitly request their children, including while routing is off. Claude's journey answers the visible permission prompt for the exact read-only orchestrator check, -verified against the native pending command. Evidence tests reject added shell +waiting for the native pending command to be flushed before confirming it. +Evidence tests reject added shell commands and other permission selections. Live automatic delegation and legacy-hook execution are not covered by these tests. `test_integration_evidence.py` checks native tool-result extraction for both agents, diff --git a/tests/integration/README.md b/tests/integration/README.md index e2f27ae51..37a1aef29 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -304,7 +304,8 @@ unverified by this suite. The Claude on/off/on journey handles the visible permission prompt for the exact read-only orchestrator check. It compares the prompt with the native pending -command before selecting the one-time Yes option, then still requires the routed +command, waiting within the existing deadline for the transcript to catch up, +before selecting the one-time Yes option. It still requires the routed banner, completed child, and correlated routing decision. The portable `../test_claude_windows_smart_routing.py` checks the Windows diff --git a/tests/integration/test_ug_smart_routing_hooks.py b/tests/integration/test_ug_smart_routing_hooks.py index f5e151bca..7ea4b0181 100644 --- a/tests/integration/test_ug_smart_routing_hooks.py +++ b/tests/integration/test_ug_smart_routing_hooks.py @@ -103,11 +103,13 @@ def routed_banner_visible(screen): for path, records in agent_sessions(session, agent).items() if not is_child_session(agent, path, records) ] - assert ( + # Claude can render the dialog before flushing its tool call to disk. + if not ( agent == "claude" and len(roots) == 1 and is_orchestrator_check_permission(screen, roots[0]) - ), "Unrecognized permission before subagent routing:\n" + screen + ): + return False tui.send("\r", "allow the read-only orchestrator routing-state check") permission_in_progress = True return False From a7a511ca0fb8fe0c5881db17c8312650494da3b4 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 17:29:35 +0000 Subject: [PATCH 07/25] Handle Claude check permissions before transcript persistence --- tests/README.md | 6 +- tests/integration/README.md | 6 +- .../test_ug_smart_routing_hooks.py | 18 +++--- tests/integration/utils/evidence.py | 35 ++++++----- tests/test_integration_evidence.py | 60 +++++++++++-------- 5 files changed, 64 insertions(+), 61 deletions(-) diff --git a/tests/README.md b/tests/README.md index 0d017dec3..5c0b11ed5 100644 --- a/tests/README.md +++ b/tests/README.md @@ -153,9 +153,9 @@ The toggle integration journeys require both bundled skills, verify the saved session controls and native tool-result confirmation after each toggle, and explicitly request their children, including while routing is off. Claude's journey answers the visible permission prompt for the exact read-only orchestrator check, -waiting for the native pending command to be flushed before confirming it. -Evidence tests reject added shell -commands and other permission selections. Live automatic +using the dialog's command because pending calls may not yet be in the transcript. +Evidence tests reject added shell commands, commands in scrollback or descriptions, +and other permission selections. Live automatic delegation and legacy-hook execution are not covered by these tests. `test_integration_evidence.py` checks native tool-result extraction for both agents, including collapsed-output records, and excludes user echoes and assistant claims. diff --git a/tests/integration/README.md b/tests/integration/README.md index 37a1aef29..f68cfa637 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -303,9 +303,9 @@ delegation and legacy-hook execution remain unverified by this suite. The Claude on/off/on journey handles the visible permission prompt for the exact -read-only orchestrator check. It compares the prompt with the native pending -command, waiting within the existing deadline for the transcript to catch up, -before selecting the one-time Yes option. It still requires the routed +read-only orchestrator check. It verifies the command in the current dialog and +waits for the dialog to settle before selecting the one-time Yes option. Pending +calls need not appear in the native transcript until approval. It still requires the routed banner, completed child, and correlated routing decision. The portable `../test_claude_windows_smart_routing.py` checks the Windows diff --git a/tests/integration/test_ug_smart_routing_hooks.py b/tests/integration/test_ug_smart_routing_hooks.py index 7ea4b0181..43eeae741 100644 --- a/tests/integration/test_ug_smart_routing_hooks.py +++ b/tests/integration/test_ug_smart_routing_hooks.py @@ -98,18 +98,14 @@ def routed_banner_visible(screen): if "Do you want to proceed?" in screen: if permission_in_progress: return False - roots = [ - records - for path, records in agent_sessions(session, agent).items() - if not is_child_session(agent, path, records) - ] - # Claude can render the dialog before flushing its tool call to disk. - if not ( - agent == "claude" - and len(roots) == 1 - and is_orchestrator_check_permission(screen, roots[0]) - ): + if agent != "claude" or not is_orchestrator_check_permission(screen): return False + # Claude briefly ignores input when a permission dialog opens. + tui.wait_for( + is_orchestrator_check_permission, + "the read-only orchestrator permission dialog to settle", + timeout=5, + ) tui.send("\r", "allow the read-only orchestrator routing-state check") permission_in_progress = True return False diff --git a/tests/integration/utils/evidence.py b/tests/integration/utils/evidence.py index 3cd988aa7..fbaa7f5d2 100644 --- a/tests/integration/utils/evidence.py +++ b/tests/integration/utils/evidence.py @@ -88,25 +88,24 @@ def is_child_session(agent: str, path: str, records: list[dict]) -> bool: return _AGENT_HELPERS.get(agent, codex).is_child_session(path, records) -def is_orchestrator_check_permission(screen: str, records: list[dict]) -> bool: - """Match the visible Claude prompt against its actual pending command.""" +def is_orchestrator_check_permission(screen: str) -> bool: + """Recognize one-time approval for the exact read-only orchestrator check.""" command = '"$UCODE_SMART_ROUTER_PYTHON" -m ucode.smart_routing.orchestrator --check' - if not ( - "Do you want to proceed?" in screen - and re.search(r"(?m)^\s*Bash command\s*$", screen) - and re.search(r"(?m)^\s*[›❯>]\s*1\.\s*Yes\s*$", screen) - and re.search(rf"(?m)^[ \t]*{re.escape(command)}[ \t]*$", screen) - ): - return False - for record in reversed(records): - if record.get("type") != "assistant": - continue - for part in reversed(record.get("message", {}).get("content", [])): - if isinstance(part, dict) and part.get("type") == "tool_use": - return ( - part.get("name") == "Bash" and part.get("input", {}).get("command") == command - ) - return False + # Pending calls need not reach the transcript until approval. Inspect the + # current dialog's first command line; multiline commands have a │ gutter. + dialog = re.split(r"(?m)^[ \t]*─{3,}[ \t]*$", screen)[-1] + return bool( + re.match( + r"\s*Bash command[ \t]*\n" + r"(?:[ \t]*Tip:[^\n]*\n)?" + r"[ \t]*\n" + rf"[ \t]*{re.escape(command)}[ \t]*\n", + dialog, + ) + and re.search(r"(?m)^[ \t]*Do you want to proceed\?[ \t]*$", dialog) + and re.search(r"(?m)^[ \t]*[›❯>][ \t]*1\.[ \t]*Yes[ \t]*$", dialog) + and re.search(r"(?m)^[ \t]*Esc to cancel\b", dialog) + ) def completed_task_models(session, agent: str, answer_value: str) -> set[str]: diff --git a/tests/test_integration_evidence.py b/tests/test_integration_evidence.py index ec793a4e6..a440af7b2 100644 --- a/tests/test_integration_evidence.py +++ b/tests/test_integration_evidence.py @@ -86,38 +86,46 @@ def test_transient_retries_and_running_tasks_are_not_terminal_errors(screen): @pytest.mark.parametrize( - "command_suffix,tool_name,selected,accepted", + "command_suffix,title,selected,accepted", [ - ("", "Bash", "1. Yes", True), - ("; echo unrelated", "Bash", "1. Yes", False), - ("\necho unrelated", "Bash", "1. Yes", False), - ("", "Read", "1. Yes", False), - ("", "Bash", "2. Yes, and switch to auto mode", False), + ("", "Bash command", "1. Yes", True), + ("; echo unrelated", "Bash command", "1. Yes", False), + ("\necho unrelated", "Bash command", "1. Yes", False), + ("", "Read file", "1. Yes", False), + ("", "Bash command", "2. Yes, and switch to auto mode", False), + ("", "Bash command", "3. No", False), ], ) -def test_orchestrator_permission_requires_exact_pending_command( - command_suffix, tool_name, selected, accepted +@pytest.mark.parametrize("tip", ["", " Tip: auto mode handles these prompts for you\n"]) +def test_orchestrator_permission_requires_exact_visible_command( + command_suffix, title, selected, accepted, tip ): command = '"$UCODE_SMART_ROUTER_PYTHON" -m ucode.smart_routing.orchestrator --check' - screen = f"Bash command\n {command}\nDo you want to proceed?\n❯ {selected}\n" - records = [ - { - "type": "assistant", - "message": { - "content": [ - {"type": "tool_use", "name": "Bash", "input": {"command": command}}, - { - "type": "tool_use", - "name": tool_name, - "input": {"command": command + command_suffix}, - }, - ] - }, - } - ] + displayed = command + command_suffix + if "\n" in displayed: + displayed = "\n".join("│ " + line for line in displayed.splitlines()) + screen = ( + "Earlier tool output\n" + "─" * 80 + "\n" + f" {title}\n{tip}\n" + f" {displayed}\n Check smart routing gate status\n\n" + f" Contains simple_expansion\n\n Do you want to proceed?\n ❯ {selected}\n\n" + " Esc to cancel · Tab to amend\n" + ) + + assert evidence.is_orchestrator_check_permission(screen) is accepted + + +@pytest.mark.parametrize("command_location", ["scrollback", "description"]) +def test_orchestrator_command_outside_dialog_command_does_not_grant_permission(command_location): + command = '"$UCODE_SMART_ROUTER_PYTHON" -m ucode.smart_routing.orchestrator --check' + screen = ( + f"{command if command_location == 'scrollback' else ''}\n" + "─" * 80 + "\n" + " Bash command\n\n echo unrelated\n" + f" {command if command_location == 'description' else 'An unrelated command'}\n\n" + " Do you want to proceed?\n ❯ 1. Yes\n\n Esc to cancel · Tab to amend\n" + ) - assert evidence.is_orchestrator_check_permission(screen, records) is accepted - assert not evidence.is_orchestrator_check_permission(screen, []) + assert not evidence.is_orchestrator_check_permission(screen) @pytest.mark.parametrize("agent", ["claude", "codex"]) From f95033333051d43e372258ed9238bffe7824de2f Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 18:09:54 +0000 Subject: [PATCH 08/25] Remove monkeypatched orchestrator test additions --- tests/README.md | 28 +- tests/integration/README.md | 21 +- tests/test_agent_codex.py | 66 ----- tests/test_claude_smart_routing_v2.py | 1 - tests/test_claude_windows_smart_routing.py | 1 - tests/test_cli.py | 28 -- tests/test_codex_config.py | 18 -- tests/test_codex_smart_routing_v2.py | 122 +-------- tests/test_orchestrator.py | 300 --------------------- tests/test_orchestrator_legacy_plugins.py | 194 ------------- tests/test_smart_router.py | 27 +- 11 files changed, 21 insertions(+), 785 deletions(-) delete mode 100644 tests/test_orchestrator.py delete mode 100644 tests/test_orchestrator_legacy_plugins.py diff --git a/tests/README.md b/tests/README.md index 5c0b11ed5..faf666608 100644 --- a/tests/README.md +++ b/tests/README.md @@ -131,35 +131,23 @@ that Claude settings and Codex's shell policy carry the interpreter and session These are component checks; they do not establish native skill permission matching or PowerShell execution. -`test_orchestrator.py` covers shared routing state, off/on transitions, suppression -of retained skills outside eligible sessions, root prompts and compaction, -user-hook preservation, and bundled Claude roles. Subprocess checks verify that -`--check` rejects ineligible sessions and leaves legacy preference files untouched, -even when they are malformed. The installed skill's check command follows live -toggles using the launching interpreter in `test_smart_router.py`. Routing checks -cover Claude/Codex role contracts without model preferences. Separate preference -editing, locking, and recovery are no longer part of the workflow. -`test_orchestrator_legacy_plugins.py` -and the Claude/Codex launcher tests check native per-launch overrides that disable -legacy marketplace registrations with routing on or off, including Codex's -app-server and remote TUI. They check config discovery, unrelated-plugin and hook -preservation, and unchanged saved settings, including nested project registrations -and `--cd`. `test_codex_config.py` covers launch-directory resolution and argument -serialization. Codex launch tests check that UG supplies only its own hooks, -preserves caller arguments and saved configuration, and starts no helper process -for subagent-only routing. Native hook merging and project trust belong to Codex; -these component assertions do not exercise its hook loader. The toggle integration journeys require both bundled skills, verify the saved session controls and native tool-result confirmation after each toggle, and explicitly request their children, including while routing is off. Claude's journey answers the visible permission prompt for the exact read-only orchestrator check, using the dialog's command because pending calls may not yet be in the transcript. Evidence tests reject added shell commands, commands in scrollback or descriptions, -and other permission selections. Live automatic -delegation and legacy-hook execution are not covered by these tests. +and other permission selections. `test_integration_evidence.py` checks native tool-result extraction for both agents, including collapsed-output records, and excludes user echoes and assistant claims. +Dedicated regression coverage is missing for root-only orchestrator activation, +compaction, retained skills in ineligible sessions, nested-session eligibility, +role-contract preservation, and isolation from legacy preference files. Legacy-plugin +suppression and preservation of saved settings and unrelated plugins/hooks are also +not covered by these journeys. Codex's native hook merging and project trust, +automatic delegation, and legacy-hook execution remain outside the integration suite. + The portable Windows routing test checks native executable forwarding, generated hooks/plugins, caller arguments, and cleanup without Unix imports. It does not establish live Windows hook execution or interactive routing. diff --git a/tests/integration/README.md b/tests/integration/README.md index f68cfa637..fbdff561b 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -287,20 +287,13 @@ the native tool-result records, and a new assistant answer after each skill invo Collapsed terminal output is allowed; the answer need not repeat the CLI's exact wording. Each following child still verifies whether a routing decision occurred. Their off-phase child is an explicit user-requested delegation; -these journeys do not establish automatic orchestration behavior. Shared on/off -state, root-only activation, compaction, retained skills, role contracts, and -isolation from legacy preferences are covered in `../test_orchestrator.py`. -`../test_smart_router.py` executes the installed routing-check command across -session toggles. Separate preference editing, locking, and recovery are not part -of the workflow. `../test_orchestrator_legacy_plugins.py` and -launcher component tests check per-launch suppression of installed legacy plugins, -including non-routed launches, nested project config and `--cd`, and preservation -of saved settings and unrelated plugins/hooks. Codex launcher component tests -check that UG supplies only its own hooks, preserves caller arguments, and starts -no helper process for subagent-only routing. Codex's native hook merging, project -trust, and execution of pre-existing hooks are not exercised by this integration suite. Live automatic -delegation and legacy-hook execution remain -unverified by this suite. +these journeys do not establish automatic orchestration behavior. Root-only +activation, compaction, retained skills in ineligible sessions, nested-session +eligibility, role-contract preservation, and isolation from legacy preference files +lack dedicated regression coverage. The journeys do not cover per-launch legacy-plugin +suppression or preservation of saved settings and unrelated plugins/hooks. +Codex's native hook merging, project trust, and execution of pre-existing hooks +are not exercised by this integration suite. The Claude on/off/on journey handles the visible permission prompt for the exact read-only orchestrator check. It verifies the command in the current dialog and diff --git a/tests/test_agent_codex.py b/tests/test_agent_codex.py index a4d7e2932..935e9712e 100644 --- a/tests/test_agent_codex.py +++ b/tests/test_agent_codex.py @@ -4,7 +4,6 @@ import json import os -import tomllib from pathlib import Path from unittest.mock import Mock @@ -994,63 +993,6 @@ def test_sets_oauth_token(self, tmp_path, monkeypatch): assert os.environ["OAUTH_TOKEN"] == "fresh-token" assert launches[0][-1] == "--search" - @pytest.mark.parametrize("version", ["0.133.0", "0.154.0"]) - def test_non_routed_launch_suppresses_legacy_plugin(self, tmp_path, monkeypatch, version): - launches = self._patch(tmp_path, monkeypatch) - monkeypatch.setattr(codex, "agent_version", lambda _binary: version) - plugin = "model-orchestrator@isaac-sync-eng-plugin-marketplace-experimental" - user_config = tmp_path / "config.toml" - user_config.write_text( - f'[plugins."{plugin}"]\nenabled = true\n' - '[plugins."unrelated@marketplace"]\nenabled = true\n' - ) - project = tmp_path / "project" - (project / ".codex").mkdir(parents=True) - project_config = project / ".codex/config.toml" - project_config.write_text( - '[plugins."model-orchestrator@project"]\nenabled = true\n' - '[plugins."unrelated@project"]\nenabled = true\n' - ) - paths = (user_config, codex.CODEX_CONFIG_PATH, project_config) - before = {path: path.read_bytes() for path in paths} - - codex.launch( - {"workspace": WS}, ["--cd", str(project), "exec", "hello"], options=LaunchOptions() - ) - - (argv,) = launches - override = next(arg for arg in argv if arg.startswith("plugins=")) - assert tomllib.loads(override) == { - "plugins": { - plugin: {"enabled": False}, - "model-orchestrator@project": {"enabled": False}, - } - } - assert argv[-4:] == ["--cd", str(project), "exec", "hello"] - assert {path: path.read_bytes() for path in paths} == before - - def test_legacy_suppression_preserves_profile_plugin_overrides(self, tmp_path, monkeypatch): - launches = self._patch(tmp_path, monkeypatch) - profile = codex.CODEX_CONFIG_PATH - profile.write_text( - profile.read_text() + '[plugins."model-orchestrator@marketplace"]\nenabled = true\n' - '[plugins."unrelated@marketplace"]\nenabled = false\n' - ) - user_config = tmp_path / "config.toml" - user_config.write_text('[plugins."unrelated@marketplace"]\nenabled = true\n') - before = user_config.read_bytes(), profile.read_bytes() - - codex.launch({"workspace": WS}, ["exec", "hello"], options=LaunchOptions()) - - (override,) = [arg for arg in launches[0] if arg.startswith("plugins=")] - assert tomllib.loads(override) == { - "plugins": { - "model-orchestrator@marketplace": {"enabled": False}, - "unrelated@marketplace": {"enabled": False}, - } - } - assert (user_config.read_bytes(), profile.read_bytes()) == before - @pytest.mark.parametrize("custom_catalog", [None, "/user/isaac-app-model-catalog.json"]) def test_native_update_detaches_catalog_without_discovery( self, tmp_path, monkeypatch, custom_catalog @@ -1544,9 +1486,6 @@ def test_catalog_cleanup_does_not_mask_write_failure(self, tmp_path, monkeypatch def test_injects_otel_config_when_tracing_enabled(self, tmp_path, monkeypatch): self._patch(tmp_path, monkeypatch) - user_config = tmp_path / "config.toml" - user_config.write_text('[plugins."model-orchestrator@marketplace"]\nenabled = true\n') - before = user_config.read_bytes() server = Mock(server_address=("127.0.0.1", 54321)) cache = Mock() client = Mock() @@ -1580,11 +1519,6 @@ def start_otel_proxy(workspace, token_provider): assert 'protocol = "binary"' in otel assert "Authorization" not in otel # no credential in argv; the proxy injects it assert argv[-2:] == ["exec", "hi"] - override = next(arg for arg in argv if arg.startswith("plugins=")) - assert tomllib.loads(override) == { - "plugins": {"model-orchestrator@marketplace": {"enabled": False}} - } - assert user_config.read_bytes() == before def test_no_otel_config_when_tracing_disabled(self, tmp_path, monkeypatch): launches = self._patch(tmp_path, monkeypatch) diff --git a/tests/test_claude_smart_routing_v2.py b/tests/test_claude_smart_routing_v2.py index 15faccc7b..1cffba3ff 100644 --- a/tests/test_claude_smart_routing_v2.py +++ b/tests/test_claude_smart_routing_v2.py @@ -483,7 +483,6 @@ def send_signal(self, _signal): assert claude_hooks.FIRST_PROMPT_SOCKET_ENV not in env # Subagent routing is fully wired; only the first-prompt machinery is absent. assert "route-first-prompt" not in str(settings["hooks"]) - assert "ucode.smart_routing.orchestrator" in str(settings["hooks"]["UserPromptSubmit"]) assert "route-subagent" in str(settings["hooks"]["PreToolUse"]) assert settings["modelOverrides"] == {"claude-opus-4-8": "system.ai.claude-opus-4-8"} assert captured["plugin_models"] == {"system.ai.claude-opus-4-8"} diff --git a/tests/test_claude_windows_smart_routing.py b/tests/test_claude_windows_smart_routing.py index 6a3807b25..62741fae5 100644 --- a/tests/test_claude_windows_smart_routing.py +++ b/tests/test_claude_windows_smart_routing.py @@ -121,7 +121,6 @@ def send_signal(self, _signal): assert settings["env"][v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR] == "1" assert v2.ENABLE_SMART_ROUTING_ENV_VAR not in settings["env"] assert "route-first-prompt" not in str(settings["hooks"]) - assert "ucode.smart_routing.orchestrator" in str(settings["hooks"]["UserPromptSubmit"]) assert "route-subagent" in str(settings["hooks"]["PreToolUse"]) assert settings["modelOverrides"] == {"claude-opus-4-8": "system.ai.claude-opus-4-8"} assert not captured["settings_path"].exists() diff --git a/tests/test_cli.py b/tests/test_cli.py index df4592e1e..7e5124b8b 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -677,34 +677,6 @@ def test_codex_and_claude_share_smart_routing_policy( assert options.launch_smart_routing is expected - @pytest.mark.parametrize("tool", ["claude", "codex"]) - @pytest.mark.parametrize("routing_enabled", [False, True]) - def test_launch_discards_parent_session_before_applying_eligibility( - self, monkeypatch, tmp_path, tool, routing_enabled - ): - from ucode.smart_routing import session_env - - parent = tmp_path / "parent-env.json" - parent.write_text("{}") - monkeypatch.setenv(session_env.SESSION_ENV_VAR, str(parent)) - monkeypatch.setenv(session_env.SESSION_PYTHON_ENV_VAR, "/parent/python") - monkeypatch.setenv("ENABLE_SMART_ROUTING_V2", "1" if routing_enabled else "0") - observed = [] - - def launch(_tool, _state, _args, *, options): - observed.append( - (options.launch_smart_routing, os.environ.get(session_env.SESSION_ENV_VAR)) - ) - - with _launch_policy_patches(None) as calls: - calls["launch"].side_effect = launch - result = runner.invoke(app, [tool]) - - assert result.exit_code == 0, result.output - assert observed == [(routing_enabled, None)] - assert os.environ[session_env.SESSION_ENV_VAR] == str(parent) - assert parent.read_text() == "{}" - @pytest.mark.parametrize( ("tool_args", "expected"), [ diff --git a/tests/test_codex_config.py b/tests/test_codex_config.py index 918d1fa09..c1fe6c5b0 100644 --- a/tests/test_codex_config.py +++ b/tests/test_codex_config.py @@ -1,9 +1,7 @@ from __future__ import annotations -import pytest import tomlkit -from ucode import codex_config from ucode.agents import codex from ucode.codex_config import codex_config_args @@ -62,19 +60,3 @@ def test_renders_nested_tables_from_parsed_profile(self): assert 'http_headers = {User-Agent = "ucode"}' in provider_override assert 'auth = {command = "ucode", args = ["codex-token"]}' in provider_override assert 'tui={model_availability_nux = {"gpt-5.6-sol" = 1}}' in args - - -@pytest.mark.parametrize( - "args", - [ - ["--cd", "project"], - ["--cd=project"], - ["-C", "project"], - ["-Cproject"], - ["-C=project"], - ["--cd", "wrong", "--cd", "project", "--", "--cd", "ignored"], - ], -) -def test_codex_working_directory(tmp_path, monkeypatch, args): - monkeypatch.chdir(tmp_path) - assert codex_config.codex_working_directory(args) == tmp_path / "project" diff --git a/tests/test_codex_smart_routing_v2.py b/tests/test_codex_smart_routing_v2.py index b3496d86b..1597a84b7 100644 --- a/tests/test_codex_smart_routing_v2.py +++ b/tests/test_codex_smart_routing_v2.py @@ -2,7 +2,6 @@ import json import os -import tomllib from types import SimpleNamespace import pytest @@ -181,26 +180,15 @@ def test_startup_config_precedence( @pytest.mark.parametrize( ("platform_name", "tui_has_provider"), [("posix", False), ("nt", True)] ) - @pytest.mark.parametrize("legacy_plugin", [False, True]) def test_owns_app_server_interposer_and_tui_lifecycle( - self, tmp_path, monkeypatch, platform_name, tui_has_provider, legacy_plugin + self, monkeypatch, platform_name, tui_has_provider ): processes = [] interposer_args = {} stopped = [] token_calls = [] monkeypatch.setenv(v2.ENABLE_SMART_ROUTING_ENV_VAR, "1") - monkeypatch.setenv("CODEX_HOME", str(tmp_path)) - user_config = tmp_path / "config.toml" - user_config.write_text( - '[plugins."unrelated@marketplace"]\nenabled = true\n' - + ( - '[plugins."model-orchestrator@marketplace"]\nenabled = true\n' - if legacy_plugin - else "" - ) - ) - before = user_config.read_bytes() + monkeypatch.setenv("CODEX_HOME", "/user/codex-home") monkeypatch.setattr(v2, "os", SimpleNamespace(name=platform_name, environ=os.environ)) monkeypatch.setattr(codex, "ug_version", lambda: "0.1.0") monkeypatch.setattr(codex, "agent_version", lambda binary: "0.148.0") @@ -284,28 +272,14 @@ def start_interposer(*args, **kwargs): "shell_environment_policy.set.UCODE_SMART_ROUTER_PYTHON=" + json.dumps(os.environ["UCODE_SMART_ROUTER_PYTHON"]) ) in config_values - assert 'shell_environment_policy.set.ENABLE_SMART_ROUTING_V2="1"' in config_values - assert "features.hooks=true" in config_values - plugin_overrides = [value for value in config_values if value.startswith("plugins=")] - if legacy_plugin: - (override,) = plugin_overrides - assert tomllib.loads(override) == { - "plugins": {"model-orchestrator@marketplace": {"enabled": False}} - } - else: - assert plugin_overrides == [] - for event in ("UserPromptSubmit", "SessionStart"): - hook = next(value for value in config_values if value.startswith(f"hooks.{event}=")) - assert "ucode.smart_routing.orchestrator" in hook assert processes[0].argv[-2:] == [ "--listen", "ws://127.0.0.1:41001", ] assert processes[0].kwargs["env"][v2.OAUTH_TOKEN_ENV_VAR] == "token-1" - assert processes[0].kwargs["env"]["CODEX_HOME"] == str(tmp_path) + assert processes[0].kwargs["env"]["CODEX_HOME"] == "/user/codex-home" tui_argv = processes[1].argv expected_tui_args = [ - *(["--config", plugin_overrides[0]] if legacy_plugin else []), "--remote", "ws://127.0.0.1:41002", "--model", @@ -336,7 +310,6 @@ def start_interposer(*args, **kwargs): assert interposer_args["kwargs"]["switch_message_fn"] is v2.format_routing_notice assert stopped == [True] assert processes[0].terminated is True - assert user_config.read_bytes() == before def test_managed_http_headers_reach_app_server_config(self, monkeypatch): # Smart routing rebuilds the overlay and passes it to the app-server as `-c` overrides that @@ -391,22 +364,16 @@ def send_signal(self, _signal): assert "x-databricks-workspace" in provider_arg assert "eng-ml-inference" in provider_arg - @pytest.mark.parametrize("legacy_plugin", [False, True]) - def test_subagent_only_launch_runs_tui_directly(self, tmp_path, monkeypatch, legacy_plugin): + def test_subagent_only_launch_runs_tui_directly(self, tmp_path, monkeypatch): monkeypatch.setenv(v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR, "1") monkeypatch.setenv("CODEX_HOME", str(tmp_path)) - user_config = tmp_path / "config.toml" - user_config.write_text( - '[plugins."model-orchestrator@marketplace"]\nenabled = true\n' if legacy_plugin else "" - ) - before = user_config.read_bytes() monkeypatch.setattr(codex, "ug_version", lambda: "0.1.0") monkeypatch.setattr(codex, "agent_version", lambda binary: "0.148.0") monkeypatch.setattr(v2, "get_databricks_token", lambda *_args, **_kwargs: "token") monkeypatch.setattr( v2.subprocess, "Popen", - lambda *_args, **_kwargs: pytest.fail("subagent-only routing starts no routing server"), + lambda *_args, **_kwargs: pytest.fail("subagent-only routing spawns no app-server"), ) monkeypatch.setattr( codex_interposer, @@ -446,21 +413,6 @@ def fake_exec(argv): "shell_environment_policy.set.UCODE_SMART_ROUTER_PYTHON=" + json.dumps(os.environ["UCODE_SMART_ROUTER_PYTHON"]) ) in argv - assert 'shell_environment_policy.set.ENABLE_SMART_ROUTING_SUBAGENT_ONLY="1"' in argv - assert "features.hooks=true" in argv - plugin_overrides = [arg for arg in argv if arg.startswith("plugins=")] - if legacy_plugin: - (override,) = plugin_overrides - assert tomllib.loads(override) == { - "plugins": {"model-orchestrator@marketplace": {"enabled": False}} - } - else: - assert plugin_overrides == [] - assert user_config.read_bytes() == before - assert any( - arg.startswith("hooks.UserPromptSubmit=") and "ucode.smart_routing.orchestrator" in arg - for arg in argv - ) # The hook subprocesses inherit the launch environment and pass the routing gate. assert os.environ[v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR] == "1" assert os.environ[v2.OAUTH_TOKEN_ENV_VAR] == "token" @@ -489,70 +441,6 @@ def test_v2_pre_tool_hook_leaves_saved_hooks_to_codex(self, tmp_path, monkeypatc assert "--model system.ai.gpt-5-6-sol" in configured[0]["hooks"][0]["command"] assert (codex_home / "config.toml").read_bytes() == before - def test_subagent_launch_leaves_native_hooks_to_codex(self, tmp_path, monkeypatch): - monkeypatch.setenv(v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR, "1") - monkeypatch.setattr(v2, "get_databricks_token", lambda *args: "token") - monkeypatch.setattr( - v2.subprocess_cross_os, - "popen", - lambda *args, **kwargs: pytest.fail("subagent-only launch needs no helper process"), - ) - user_home = tmp_path / "user" - user_home.mkdir() - monkeypatch.setenv("CODEX_HOME", str(user_home)) - project = tmp_path / "project" - (project / ".codex").mkdir(parents=True) - user_config = user_home / "config.toml" - project_config = project / ".codex/config.toml" - for source, path in (("user", user_config), ("project", project_config)): - path.write_text( - "".join( - f"[[hooks.{event}]]\n[[hooks.{event}.hooks]]\n" - f'type = "command"\ncommand = "{source}-{event}"\n' - for event in ("PreToolUse", "UserPromptSubmit", "SessionStart", "Stop") - ) - + '[plugins."model-orchestrator@project"]\nenabled = true\n' - ) - before = {path: path.read_bytes() for path in (user_config, project_config)} - launches = [] - caller_config = ["--config", 'projects={"/other"={trust_level="untrusted"}}'] - tool_args = [*caller_config, "--cd", str(project)] - - def launch(argv): - launches.append(argv) - raise SystemExit(0) - - monkeypatch.setattr(v2, "exec_or_spawn", launch) - with pytest.raises(SystemExit): - v2.launch_codex( - {"workspace": WS, "codex_models": ["gpt-5.6-sol"]}, - tool_args, - binary="/selected/codex", - start_model="gpt-5.6-sol", - render_overlay=lambda *args, **kwargs: {"model": "gpt-5.6-sol"}, - ) - - (argv,) = launches - expected_handlers = { - "PreToolUse": "codex-router-hook route-subagent", - "UserPromptSubmit": "ucode.smart_routing.orchestrator", - "SessionStart": "ucode.smart_routing.orchestrator", - } - assert {arg.split("=", 1)[0] for arg in argv if arg.startswith("hooks.")} == { - f"hooks.{event}" for event in expected_handlers - } - for event, handler in expected_handlers.items(): - value = next(arg for arg in argv if arg.startswith(f"hooks.{event}=")) - (group,) = tomllib.loads(value)["hooks"][event] - (hook,) = group["hooks"] - assert handler in hook["command"] - plugin_arg = next(arg for arg in argv if arg.startswith("plugins=")) - assert tomllib.loads(plugin_arg) == { - "plugins": {"model-orchestrator@project": {"enabled": False}} - } - assert argv[-len(tool_args) :] == tool_args - assert all(path.read_bytes() == content for path, content in before.items()) - def test_v2_pre_tool_hook_uses_current_model(self, tmp_path, monkeypatch): monkeypatch.setattr("ucode.databricks.ug_binary", lambda: "/bin/ug") codex_home = tmp_path / ".codex" diff --git a/tests/test_orchestrator.py b/tests/test_orchestrator.py deleted file mode 100644 index 928ac5804..000000000 --- a/tests/test_orchestrator.py +++ /dev/null @@ -1,300 +0,0 @@ -"""Orchestration follows the same launch eligibility and live state as routing.""" - -import copy -import io -import json -import os -import shlex -import subprocess -import sys -from pathlib import Path - -import pytest -from typer.testing import CliRunner - -from ucode import cli, skills -from ucode.smart_routing import codex_routing, orchestrator, routing, session_env, v2 - - -@pytest.fixture -def routed_session(monkeypatch): - monkeypatch.setenv(v2.ENABLE_SMART_ROUTING_ENV_VAR, "1") - monkeypatch.delenv(v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR, raising=False) - monkeypatch.delenv("ISAAC_LAUNCH_MODE", raising=False) - monkeypatch.setattr(skills, "_skills_source", lambda: Path(__file__).parents[1] / "skills") - return session_env.start_session() - - -@pytest.mark.parametrize("agent", ["claude", "codex"]) -def test_toggle_supersedes_workflow_and_reenables_it(routed_session, agent): - payload = {"hook_event_name": "UserPromptSubmit"} - workflow = orchestrator.hook_output(payload)["hookSpecificOutput"]["additionalContext"] - assert "Apply the UG model-orchestrator workflow" in workflow - runner = CliRunner() - - off = runner.invoke(cli.app, [agent, "--disable-smart-routing"]) - assert off.exit_code == 0, off.output - assert "do not start new automatic" in " ".join(off.output.split()) - assert not v2.smart_routing_enabled(session_env.effective_environment()) - assert not orchestrator.enabled() - assert orchestrator.hook_output(payload)["hookSpecificOutput"]["additionalContext"] == ( - orchestrator.DISABLED_CONTEXT - ) - with pytest.raises(ValueError, match="supersedes any earlier"): - orchestrator.require_enabled() - - on = runner.invoke(cli.app, [agent, "--enable-smart-routing"]) - assert on.exit_code == 0, on.output - assert "Automatic orchestration is on" in on.output - assert v2.smart_routing_enabled(session_env.effective_environment()) - assert orchestrator.hook_output(payload)["hookSpecificOutput"]["additionalContext"] == workflow - - -@pytest.mark.parametrize("mode", ["no-session", "missing-session", "off", "omni"]) -def test_retained_skill_cannot_enable_orchestration(routed_session, monkeypatch, mode): - if mode == "no-session": - monkeypatch.delenv(session_env.SESSION_ENV_VAR) - elif mode == "missing-session": - routed_session.unlink() - elif mode == "off": - session_env.set_session_environment(dict.fromkeys(v2.SMART_ROUTING_ENV_KEYS, "0")) - else: - monkeypatch.setenv("ISAAC_LAUNCH_MODE", " OMNI ") - assert (orchestrator.skill_directory() / "SKILL.md").is_file() - assert not orchestrator.enabled() - with pytest.raises(ValueError, match="do not start new automatic delegation"): - orchestrator.require_enabled() - result = subprocess.run( - [sys.executable, "-m", orchestrator.HOOK_MODULE, "--check"], - stdin=subprocess.DEVNULL, - capture_output=True, - text=True, - timeout=20, - ) - assert result.returncode == 1 - assert result.stdout == "" - assert result.stderr == orchestrator.DISABLED_CONTEXT + "\n" - - -def test_check_does_not_read_or_modify_legacy_preferences(routed_session, tmp_path): - legacy_files = { - ".model-orchestrator.json": b"{invalid project preferences", - ".model-orchestrator.json.transaction.json": b"{unfinished update", - ".config/model-orchestrator/config.json": b"{invalid user preferences", - ".config/model-orchestrator/config.json.lock": b"existing lock file", - ".claude/agents/model-orchestrator-custom-project-reviewer.md": b"user-edited agent", - ".codex/config.toml": b"[invalid catalog config", - } - for relative, content in legacy_files.items(): - path = tmp_path / relative - path.parent.mkdir(parents=True, exist_ok=True) - path.write_bytes(content) - before = {path: path.read_bytes() for path in tmp_path.rglob("*") if path.is_file()} - - result = subprocess.run( - [sys.executable, "-m", orchestrator.HOOK_MODULE, "--check"], - cwd=tmp_path, - env={ - **os.environ, - "HOME": str(tmp_path), - "USERPROFILE": str(tmp_path), - "XDG_CONFIG_HOME": str(tmp_path / ".config"), - "CLAUDE_CONFIG_DIR": str(tmp_path / ".claude"), - "CODEX_HOME": str(tmp_path / ".codex"), - }, - stdin=subprocess.DEVNULL, - capture_output=True, - text=True, - timeout=20, - ) - - assert result.returncode == 0, result.stderr - assert result.stdout == result.stderr == "" - assert {path: path.read_bytes() for path in tmp_path.rglob("*") if path.is_file()} == before - - -@pytest.mark.parametrize( - "flags", - [ - {v2.ENABLE_SMART_ROUTING_ENV_VAR: "1"}, - {v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1"}, - {v2.ENABLE_SMART_ROUTING_ENV_VAR: "0", v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR: "1"}, - {v2.ENABLE_SMART_ROUTING_ENV_VAR: "0"}, - {}, - ], -) -def test_full_and_subagent_routing_share_the_gate(routed_session, flags): - env = {session_env.SESSION_ENV_VAR: str(routed_session), **flags} - assert orchestrator.enabled(env) == v2.smart_routing_enabled( - session_env.effective_environment(env) - ) - - -def test_nested_launch_does_not_inherit_eligibility(routed_session): - before = os.environ[session_env.SESSION_ENV_VAR] - with session_env.fresh_launch(): - assert session_env.SESSION_PYTHON_ENV_VAR not in os.environ - assert not orchestrator.enabled() - # A newly eligible launch creates its own session instead of sharing the parent's toggle. - child_session = session_env.start_session() - assert child_session != routed_session - assert orchestrator.enabled() - session_env.set_session_environment(dict.fromkeys(v2.SMART_ROUTING_ENV_KEYS, "0")) - assert os.environ[session_env.SESSION_ENV_VAR] == before - assert orchestrator.enabled() - - -@pytest.mark.parametrize( - "payload,active", - [ - ({"hook_event_name": "UserPromptSubmit"}, True), - ({"hook_event_name": "UserPromptSubmit", "agent_type": "custom-root"}, True), - ({"hook_event_name": "SessionStart", "source": "compact"}, True), - ({"hook_event_name": "SessionStart", "source": "startup"}, False), - ({"hook_event_name": "UserPromptSubmit", "agent_id": "child"}, False), - ({"hook_event_name": "SessionStart", "source": "compact", "agent_id": "child"}, False), - ({"hook_event_name": []}, False), - ({"hook_event_name": "Unknown"}, False), - ([], False), - (None, False), - ], -) -def test_only_root_prompt_and_compaction_load_workflow(routed_session, payload, active): - output = orchestrator.hook_output(payload) - if not active: - assert output is None - return - context = output["hookSpecificOutput"]["additionalContext"] - directory = orchestrator.skill_directory() - assert str(directory) in context - assert context.endswith((directory / "SKILL.md").read_text()) - - -@pytest.mark.parametrize("payload", ["", "{", "null", "[]", '"text"', "{}"]) -def test_hook_entry_point_ignores_invalid_payload(monkeypatch, capsys, payload): - monkeypatch.setattr(sys, "stdin", io.StringIO(payload)) - orchestrator.main([]) - assert capsys.readouterr().out == "" - - -@pytest.mark.parametrize("state", ["missing", "directory", "invalid-encoding"]) -def test_unreadable_workflow_is_not_injected(routed_session, tmp_path, monkeypatch, state): - monkeypatch.setattr(orchestrator, "skill_directory", lambda: tmp_path) - skill = tmp_path / "SKILL.md" - if state == "directory": - skill.mkdir() - elif state == "invalid-encoding": - skill.write_bytes(b"\xff") - assert orchestrator.hook_output({"hook_event_name": "UserPromptSubmit"}) is None - - -def test_hooks_preserve_user_handlers_and_replace_only_ug_handlers(monkeypatch): - monkeypatch.setattr(sys, "executable", "/launch installation/bin/python") - user = {"hooks": [{"type": "command", "command": "user-policy"}]} - doc = { - "hooks": { - "UserPromptSubmit": [copy.deepcopy(user)], - "SessionStart": [copy.deepcopy(user)], - "PreToolUse": [copy.deepcopy(user)], - } - } - orchestrator.sync_hooks(doc, agent="codex") - first = copy.deepcopy(doc) - orchestrator.sync_hooks(doc, agent="codex") - assert doc == first - assert doc["hooks"]["PreToolUse"] == [user] - for event in ("UserPromptSubmit", "SessionStart"): - groups = doc["hooks"][event] - assert len(groups) == 2 - assert groups[0] == user - hook = groups[1]["hooks"][0] - assert shlex.split(hook["command"]) == [sys.executable, "-m", orchestrator.HOOK_MODULE] - assert hook["command_windows"] == subprocess.list2cmdline( - [sys.executable, "-m", orchestrator.HOOK_MODULE] - ) - assert doc["hooks"]["SessionStart"][1]["matcher"] == "compact" - - -def test_codex_launch_adds_only_its_prompt_and_compaction_hooks(tmp_path, monkeypatch): - config = tmp_path / "config.toml" - config.write_text( - '[[hooks.UserPromptSubmit]]\n[[hooks.UserPromptSubmit.hooks]]\ncommand = "user-prompt"\n' - '[[hooks.SessionStart]]\nmatcher = "compact"\n' - '[[hooks.SessionStart.hooks]]\ncommand = "user-compact"\n' - ) - before = config.read_bytes() - monkeypatch.setenv("CODEX_HOME", str(tmp_path)) - hooks = v2._v2_hooks({"workspace": "https://example.com"}, ["gpt-6-sol"]) - for event in ("UserPromptSubmit", "SessionStart"): - (group,) = hooks[event] - (hook,) = group["hooks"] - assert orchestrator.HOOK_MODULE in hook["command"] - assert config.read_bytes() == before - - -def test_routed_claude_plugin_contains_unchanged_roles(routed_session, tmp_path): - v2._write_routed_claude_plugin(tmp_path, ["system.ai.claude-sonnet-4-6"]) - manifest = json.loads((tmp_path / ".claude-plugin/plugin.json").read_text()) - assert manifest["name"] == "ug-smart-router" - templates = list((orchestrator.skill_directory() / "agents").glob("*.md")) - assert {path.stem for path in templates} == { - "explorer", - "researcher", - "worker", - "tester", - "reviewer", - } - for template in templates: - assert (tmp_path / "agents" / template.name).read_bytes() == template.read_bytes() - - -def test_role_contract_survives_claude_model_routing(monkeypatch): - monkeypatch.setattr( - v2, - "_request_claude_routing_decision", - lambda *_args: ( - routing.RoutingDecision( - model="system.ai.claude-sonnet-4-6", raw_model="claude-sonnet-4-6" - ), - None, - ), - ) - contract = ( - "Act as the researcher. Verify the API using the connected documentation tool. " - "Do not edit files or delegate. Stop on auth failure. Return source links and evidence." - ) - payload = { - "tool_name": "Agent", - "tool_input": {"subagent_type": "ug-smart-router:researcher", "prompt": contract}, - } - result = v2.route_claude_pre_tool_use( - payload, - workspace="https://example.com", - token="token", - available_models=["system.ai.claude-sonnet-4-6"], - ) - updated = result["hookSpecificOutput"]["updatedInput"] - assert updated["prompt"] == contract - assert updated["subagent_type"].startswith("ug-smart-router:ucode-route-") - assert "model" not in updated - - -def test_codex_routes_role_without_model_preferences(monkeypatch): - # Keep the routing path real; replace only the external selection request. - monkeypatch.setattr( - codex_routing, - "request_routing_decision", - lambda *_args, **_kwargs: ( - routing.RoutingDecision(model="system.ai.gpt-5-6-sol", raw_model="gpt-5-6-sol"), - None, - ), - ) - contract = "Act as the reviewer. Inspect this diff without editing. Report concrete bugs." - task = {"task_name": "reviewer", "message": contract, "fork_turns": "none"} - result = codex_routing.route_pre_tool_use( - {"tool_name": "collaboration.spawn_agent", "tool_input": task}, - workspace="https://example.com", - token="token", - available_models=["system.ai.gpt-5-6-sol"], - ) - assert result["hookSpecificOutput"]["updatedInput"] == {**task, "model": "gpt-5.6-sol"} diff --git a/tests/test_orchestrator_legacy_plugins.py b/tests/test_orchestrator_legacy_plugins.py deleted file mode 100644 index a5761025d..000000000 --- a/tests/test_orchestrator_legacy_plugins.py +++ /dev/null @@ -1,194 +0,0 @@ -"""The marketplace predecessor must not bypass UG's shared routing controls.""" - -import json -import tomllib - -import pytest -import tomlkit - -from ucode import codex_config -from ucode.agents import claude -from ucode.smart_routing import orchestrator, v2 - - -@pytest.mark.parametrize("launch_mode", ["direct", "relayed", "routing"]) -@pytest.mark.parametrize("routing_enabled", ["0", "1"]) -def test_claude_suppresses_legacy_plugins_without_changing_saved_settings( - tmp_path, monkeypatch, launch_mode, routing_enabled -): - monkeypatch.setenv(v2.ENABLE_SMART_ROUTING_ENV_VAR, routing_enabled) - user_plugin = "model-orchestrator@eng-plugin-marketplace-experimental" - project_plugin = "model-orchestrator@project-marketplace" - caller_plugin = "model-orchestrator@caller-marketplace" - user_settings = claude.CLAUDE_USER_SETTINGS_PATH - registry = user_settings.parent / "plugins" / "installed_plugins.json" - registry.parent.mkdir(parents=True) - registry.write_text( - json.dumps({"version": 2, "plugins": {project_plugin: [{"scope": "project"}]}}) - ) - user_settings.write_text( - json.dumps({"enabledPlugins": {user_plugin: True, "unrelated@marketplace": True}}) - ) - gateway_settings = tmp_path / "ucode-settings.json" - gateway_settings.write_text(json.dumps({"apiKeyHelper": "gateway-helper"})) - monkeypatch.setattr(claude, "CLAUDE_SETTINGS_PATH", gateway_settings) - user_hook = {"hooks": [{"type": "command", "command": "user-policy"}]} - caller = tmp_path / "caller-settings.json" - caller.write_text( - json.dumps( - { - "enabledPlugins": { - caller_plugin: True, - "unrelated@marketplace": True, - "model-orchestrator-extra@marketplace": True, - }, - "hooks": {"UserPromptSubmit": [user_hook]}, - } - ) - ) - before = { - path: path.read_bytes() for path in (user_settings, registry, gateway_settings, caller) - } - args = ["--settings", str(caller), "--print", "hello"] - - if launch_mode == "routing": - composed, remaining = claude._compose_v2_settings(args) - assert composed["enabledPlugins"][user_plugin] is False - argv = claude._build_claude_argv("claude", remaining, settings_override=composed) - else: - argv = claude._build_claude_argv("claude", args, relayed=launch_mode == "relayed") - - settings = json.loads(argv[argv.index("--settings") + 1]) - assert settings["enabledPlugins"] == { - user_plugin: False, - project_plugin: False, - caller_plugin: False, - "unrelated@marketplace": True, - "model-orchestrator-extra@marketplace": True, - } - assert settings["apiKeyHelper"] == "gateway-helper" - assert settings["hooks"]["UserPromptSubmit"] == [user_hook] - assert argv[-2:] == ["--print", "hello"] - assert args == ["--settings", str(caller), "--print", "hello"] - assert {path: path.read_bytes() for path in before} == before - - -@pytest.mark.parametrize("custom_home", [False, True]) -def test_claude_suppresses_installed_plugin_without_caller_settings( - tmp_path, monkeypatch, custom_home -): - config_dir = ( - tmp_path / "custom-claude" if custom_home else claude.CLAUDE_USER_SETTINGS_PATH.parent - ) - registry = config_dir / "plugins" / "installed_plugins.json" - registry.parent.mkdir(parents=True) - name = "model-orchestrator@custom-marketplace" - registry.write_text(json.dumps({"plugins": {name: [{"scope": "user"}]}})) - if custom_home: - monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(config_dir)) - gateway_settings = tmp_path / "ucode-settings.json" - gateway_settings.write_text("{}") - monkeypatch.setattr(claude, "CLAUDE_SETTINGS_PATH", gateway_settings) - - argv = claude._build_claude_argv("claude", ["--print", "hello"]) - - settings = json.loads(argv[argv.index("--settings") + 1]) - assert settings == {"enabledPlugins": {name: False}} - assert gateway_settings.read_text() == "{}" - - -def test_claude_without_legacy_plugin_keeps_settings_file_argument(tmp_path, monkeypatch): - user_settings = claude.CLAUDE_USER_SETTINGS_PATH - user_settings.parent.mkdir(parents=True) - user_settings.write_text( - json.dumps({"enabledPlugins": {"model-orchestrator-extra@marketplace": True}}) - ) - gateway_settings = tmp_path / "ucode-settings.json" - gateway_settings.write_text("{}") - monkeypatch.setattr(claude, "CLAUDE_SETTINGS_PATH", gateway_settings) - - assert claude._build_claude_argv("claude", ["--print", "hello"]) == [ - "claude", - "--settings", - str(gateway_settings), - "--print", - "hello", - ] - - -@pytest.mark.parametrize("custom_home", [False, True]) -def test_codex_suppresses_legacy_plugins_from_saved_config_layers( - tmp_path, monkeypatch, custom_home -): - default_profile = codex_config.DEFAULT_CODEX_CONFIG_PATH - profile = default_profile - if custom_home: - monkeypatch.setenv("CODEX_HOME", str(tmp_path / "custom-codex")) - profile = tmp_path / "custom-codex" / "ucode.config.toml" - default_profile.parent.mkdir(parents=True) - default_profile.write_text('[plugins."model-orchestrator@unused-home"]\nenabled = true\n') - managed = tmp_path / "managed_config.toml" - monkeypatch.setattr(codex_config, "codex_managed_config_path", lambda: managed) - names = { - managed: "model-orchestrator@managed-marketplace", - profile: "model-orchestrator@profile-marketplace", - profile.parent / "config.toml": "model-orchestrator@isaac-sync-user.marketplace", - } - for path, name in names.items(): - path.parent.mkdir(parents=True, exist_ok=True) - path.write_text( - tomlkit.dumps( - { - "plugins": { - name: {"enabled": True}, - "unrelated@marketplace": {"enabled": True}, - "model-orchestrator-extra@marketplace": {"enabled": True}, - } - } - ) - ) - before = {path: path.read_bytes() for path in names} - - override = orchestrator.legacy_codex_plugin_config(default_profile) - - assert override == {"plugins": {name: {"enabled": False} for name in names.values()}} - flag, value = codex_config.codex_config_args(override) - assert flag == "--config" - assert value.startswith("plugins={") - assert tomllib.loads(value) == override - assert {path: path.read_bytes() for path in before} == before - - -def test_codex_suppresses_project_plugins_from_nested_directory(tmp_path, monkeypatch): - monkeypatch.setenv("CODEX_HOME", str(tmp_path / "user")) - project = tmp_path / "project" - nested = project / "nested" - names = { - project / ".codex/config.toml": "model-orchestrator@project", - nested / ".codex/config.toml": "model-orchestrator@nested", - } - for path, name in names.items(): - path.parent.mkdir(parents=True) - path.write_text( - tomlkit.dumps( - { - "plugins": {name: {"enabled": True}, "unrelated@project": {"enabled": True}}, - "hooks": {"UserPromptSubmit": [{"hooks": [{"command": "project-policy"}]}]}, - } - ) - ) - before = {path: path.read_bytes() for path in names} - - override = orchestrator.legacy_codex_plugin_config(cwd=nested) - - assert override == {"plugins": {name: {"enabled": False} for name in names.values()}} - assert {path: path.read_bytes() for path in before} == before - - -def test_codex_without_legacy_plugin_has_no_override(tmp_path, monkeypatch): - monkeypatch.setenv("CODEX_HOME", str(tmp_path)) - (tmp_path / "config.toml").write_text( - '[plugins."model-orchestrator-extra@marketplace"]\nenabled = true\n' - ) - - assert orchestrator.legacy_codex_plugin_config() == {} diff --git a/tests/test_smart_router.py b/tests/test_smart_router.py index a5bcee6ed..81613c882 100644 --- a/tests/test_smart_router.py +++ b/tests/test_smart_router.py @@ -12,7 +12,7 @@ from typer.testing import CliRunner from ucode import cli, config_io, skills -from ucode.skills import ORCHESTRATOR_SKILL, SMART_ROUTER_SKILL +from ucode.skills import SMART_ROUTER_SKILL from ucode.smart_routing import session_env, v2 runner = CliRunner() @@ -33,8 +33,6 @@ def test_smart_routed_session_installs_skill(tmp_path, monkeypatch, agent): home = config_io.APP_DIR.parent assert home.joinpath(f".{agent}/skills/{SMART_ROUTER_SKILL}/SKILL.md").is_file() - assert home.joinpath(f".{agent}/skills/{ORCHESTRATOR_SKILL}/SKILL.md").is_file() - assert home.joinpath(f".{agent}/skills/{ORCHESTRATOR_SKILL}/agents/reviewer.md").is_file() assert not home.joinpath(f".agents/skills/{SMART_ROUTER_SKILL}").exists() assert session_path == Path(os.environ[session_env.SESSION_ENV_VAR]) assert Path(os.environ[session_env.SESSION_ENV_VAR]).is_file() @@ -44,10 +42,6 @@ def test_smart_routed_session_installs_skill(tmp_path, monkeypatch, agent): @pytest.mark.skipif(os.name == "nt", reason="Exercises the skill's POSIX shell commands") @pytest.mark.parametrize("agent", ["claude", "codex"]) def test_skill_toggles_with_launch_installation_despite_shadowed_path(tmp_path, monkeypatch, agent): - # Launch preparation is reached only after routing eligibility has passed. - monkeypatch.setenv(v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR, "1") - monkeypatch.delenv(v2.ENABLE_SMART_ROUTING_ENV_VAR, raising=False) - monkeypatch.delenv("ISAAC_LAUNCH_MODE", raising=False) # Keep the venv path (including spaces), not its resolved system Python symlink. installation = tmp_path / "launch installation" installation.symlink_to(sys.prefix, target_is_directory=True) @@ -62,10 +56,6 @@ def test_skill_toggles_with_launch_installation_despite_shadowed_path(tmp_path, session_path = v2._prepare_smart_router_session(agent) skill = config_io.APP_DIR.parent / f".{agent}/skills/{SMART_ROUTER_SKILL}/SKILL.md" commands = re.findall(r"`([^`\n]*--(?:enable|disable)-smart-routing)`", skill.read_text()) - orchestrate = config_io.APP_DIR.parent / f".{agent}/skills/{ORCHESTRATOR_SKILL}/SKILL.md" - check_command = re.search( - r'^"\$UCODE_SMART_ROUTER_PYTHON" -m [^\n]+ --check$', orchestrate.read_text(), re.M - ).group() for action in ("disable", "enable"): command = next(cmd for cmd in commands if f" {agent} --{action}-" in cmd) @@ -82,21 +72,6 @@ def test_skill_toggles_with_launch_installation_despite_shadowed_path(tmp_path, assert f"Smart Router is {'off' if action == 'disable' else 'on'}" in result.stdout expected = dict.fromkeys(v2.SMART_ROUTING_ENV_KEYS, "0") if action == "disable" else {} assert json.loads(session_path.read_text()) == expected - check = subprocess.run( - ["/bin/sh", "-c", check_command], - cwd=tmp_path, - env={**os.environ, "NO_COLOR": "1"}, - stdin=subprocess.DEVNULL, - capture_output=True, - text=True, - timeout=20, - ) - assert check.returncode == (1 if action == "disable" else 0), check.stderr - assert check.stdout == "" - if action == "disable": - assert "smart routing is off" in check.stderr - else: - assert check.stderr == "" def test_launcher_flags_control_routing_hook(tmp_path, monkeypatch): From 480c4b5bbd9712e306a80d12c51b6e364f4399b3 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 18:11:17 +0000 Subject: [PATCH 09/25] Remove standalone orchestrator plugin compatibility --- AGENTS.md | 2 - README.md | 11 ++--- skills/orchestrate/README.md | 11 +---- src/ucode/agents/claude.py | 12 +++--- src/ucode/agents/codex.py | 9 +--- src/ucode/codex_config.py | 18 -------- src/ucode/smart_routing/orchestrator.py | 55 +------------------------ src/ucode/smart_routing/v2.py | 16 +------ tests/README.md | 7 ++-- tests/conftest.py | 8 ---- tests/integration/README.md | 6 +-- tests/test_agent_claude.py | 5 +-- tests/test_managed_files.py | 4 +- 13 files changed, 19 insertions(+), 145 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index e219fb94e..bc03fddd7 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -71,7 +71,6 @@ Fields live in `~/.claude/ucode-settings.json` and the OS-managed settings file | `managedMcpServers` | Ignore | Merge | Add/update the config's MCP server entries; other entries left alone | | Smart-routing hooks | Merge | Merge | `PreToolUse`, `SessionStart`, `SubagentStart`; only `ug`'s own marked handlers, other hooks left alone | | Orchestrator hooks | Merge | Merge | Launch-only `UserPromptSubmit` and compact `SessionStart` handlers for smart-routed sessions; read the same session controls as routing | -| `enabledPlugins["model-orchestrator@…"]` | Merge | Merge | Launch-only `false` overrides for installed legacy plugins, even when smart routing is off; saved settings and other plugins left alone | @@ -89,6 +88,5 @@ Fields live in `~/.codex/ucode.config.toml` and `/etc/codex/managed_config.toml` | `model_catalog_json` | Create/replace | Create/replace | In `~/.codex/config.toml`; `ug`'s own catalog reference, for a static model list | | `mcp_servers` | Ignore | Merge | Managed file; add/update the config's MCP server entries, other entries left alone | | Smart-routing and orchestrator hooks | Merge | Merge | Launch-only `PreToolUse`, `UserPromptSubmit`, and `SessionStart` handlers; Codex combines them with its native hook sources and applies project trust; `features.hooks` is enabled for a smart-routed launch | -| `plugins."model-orchestrator@…".enabled` | Merge | Merge | Launch-only `false` overrides for legacy registrations, including Isaac-synced and project plugins, even when smart routing is off; saved settings and other plugins left alone | diff --git a/README.md b/README.md index 40d8d514a..8573808b4 100644 --- a/README.md +++ b/README.md @@ -262,15 +262,10 @@ harness behavior while routing is off. Stored skill files do not activate it in later non-routed sessions. The existing Isaac pilot gate and UG launch exclusions are unchanged. -UG automatically suppresses installed standalone `model-orchestrator` plugins, -including Isaac-synced Codex registrations, for each Claude or Codex launch. -This also applies when smart routing is off, so the old hooks cannot activate -orchestration independently. Codex project registrations are included. UG supplies -only its own hooks; Codex combines them with existing hooks and applies project -trust. Saved plugin settings and unrelated plugins -and hooks remain intact. Smart routing selects subagent models; separate role-model +UG supplies its own hooks; Codex combines them with existing hooks and applies +project trust. Smart routing selects subagent models; separate role-model preferences are ignored and their files are left untouched. See the bundled -[orchestrator documentation](skills/orchestrate/README.md) for cutover details. +[orchestrator documentation](skills/orchestrate/README.md) for details. ## Managed Files diff --git a/skills/orchestrate/README.md b/skills/orchestrate/README.md index afdf70939..7e962944c 100644 --- a/skills/orchestrate/README.md +++ b/skills/orchestrate/README.md @@ -20,16 +20,7 @@ Role instructions belong in each task prompt because routing may replace the requested Claude role or Codex model. Hook approval in the native `/hooks` UI is still required where the harness prompts for it. -## Existing installations and preferences - -UG suppresses installed `model-orchestrator` marketplace plugins for every Claude -and Codex launch, including Isaac-synced Codex registrations and launches with -smart routing off. The old activation hook does not check routing state, so its -plugin is disabled through native per-launch settings. Saved registrations, -unrelated plugins and hooks, and launches outside UG are unaffected. - -This covers marketplace installations; manually copied activation hooks or -development copies passed through `--plugin-dir` need to be removed separately. +## Routing check Separate role-model preferences are not used by the UG workflow. Existing `.model-orchestrator.json` project preferences, diff --git a/src/ucode/agents/claude.py b/src/ucode/agents/claude.py index 5b459f9cc..68e3f1d00 100644 --- a/src/ucode/agents/claude.py +++ b/src/ucode/agents/claude.py @@ -82,7 +82,6 @@ external_provider_selected, ) from ucode.os_compatibility import subprocess_cross_os -from ucode.smart_routing import orchestrator from ucode.smart_routing import v2 as smart_routing_v2 from ucode.smart_routing.claude_hooks import ( FIRST_PROMPT_SOCKET_ENV, @@ -1858,9 +1857,7 @@ def _compose_v2_settings(tool_args: list[str]) -> tuple[dict, list[str]]: settings: dict = {} for value in caller_values: settings = _merge_claude_settings(settings, _load_caller_settings(value)) - settings = _merge_claude_settings(settings, read_json_safe(CLAUDE_SETTINGS_PATH)) - orchestrator.suppress_legacy_claude_plugin(settings, CLAUDE_USER_SETTINGS_PATH) - return settings, remaining + return _merge_claude_settings(settings, read_json_safe(CLAUDE_SETTINGS_PATH)), remaining def _launch_model_args(tool_args: list[str], launch_model: str | None) -> list[str]: @@ -1975,6 +1972,10 @@ def _build_claude_argv( """ source_args = ["--setting-sources", _RELAYED_SETTING_SOURCES] if relayed else [] caller_values, remaining = _extract_caller_settings(tool_args) + if not caller_values and settings_override is None: + # No caller --settings: hand Claude ucode's settings file directly (the + # common path; behavior unchanged). + return [binary, *source_args, "--settings", str(CLAUDE_SETTINGS_PATH), *tool_args] caller_settings: dict = {} for value in caller_values: caller_settings = _merge_claude_settings(caller_settings, _load_caller_settings(value)) @@ -1983,9 +1984,6 @@ def _build_claude_argv( merged = _merge_claude_settings(caller_settings, read_json_safe(CLAUDE_SETTINGS_PATH)) if settings_override is not None: merged = _merge_claude_settings(merged, settings_override) - suppressed = orchestrator.suppress_legacy_claude_plugin(merged, CLAUDE_USER_SETTINGS_PATH) - if not caller_values and settings_override is None and not suppressed: - return [binary, *source_args, "--settings", str(CLAUDE_SETTINGS_PATH), *tool_args] merged_env = merged.get("env") if isinstance(merged_env, dict): merged_env.pop("CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY", None) diff --git a/src/ucode/agents/codex.py b/src/ucode/agents/codex.py index cadbcf09a..3c04a93ff 100644 --- a/src/ucode/agents/codex.py +++ b/src/ucode/agents/codex.py @@ -21,7 +21,6 @@ codex_config_args, codex_config_precedence_paths, codex_managed_config_path, - codex_working_directory, custom_catalog_models, ) from ucode.config_io import ( @@ -71,7 +70,6 @@ revert_managed_file, ) from ucode.os_compatibility import subprocess_cross_os -from ucode.smart_routing import orchestrator from ucode.smart_routing import v2 as smart_routing_v2 from ucode.smart_routing.codex_hooks import ( remove_smart_routing_hooks, @@ -1155,9 +1153,6 @@ def launch( if workspace: token = _launch_token(state, workspace) os.environ["OAUTH_TOKEN"] = token - legacy_plugin_config = orchestrator.legacy_codex_plugin_config( - CODEX_CONFIG_PATH, cwd=codex_working_directory(tool_args) - ) if _use_legacy_layout(): print_warning_err( f"Codex {agent_version(binary)} is outdated. Upgrade Codex to " @@ -1166,7 +1161,7 @@ def launch( ) _run_codex( state, - [binary, "--profile", CODEX_PROFILE_NAME, *codex_config_args(legacy_plugin_config)], + [binary, "--profile", CODEX_PROFILE_NAME], tool_args, otel_tracing=otel_tracing, workspace=workspace, @@ -1182,8 +1177,6 @@ def launch( f"Cannot launch Codex with the ucode profile because {CODEX_CONFIG_PATH} " "is missing or empty. Run `ucode configure --agents codex` first." ) - # Repeated CLI keys replace earlier overrides, so preserve the profile's other plugins here. - deep_merge_dict(profile_doc, legacy_plugin_config) _set_provider_header(profile_doc, provider) _set_parent_schema_header(profile_doc, parent_schema if not provider else None) updating = tool_args[:1] == ["update"] diff --git a/src/ucode/codex_config.py b/src/ucode/codex_config.py index c144d6af3..b93148f5d 100644 --- a/src/ucode/codex_config.py +++ b/src/ucode/codex_config.py @@ -18,24 +18,6 @@ DEFAULT_CODEX_CONFIG_PATH = Path.home() / ".codex" / f"{CODEX_PROFILE_NAME}.config.toml" -def codex_working_directory(tool_args: list[str]) -> Path: - """Resolve the launch directory before looking up project configuration.""" - directory = Path.cwd() - args = iter(tool_args) - for arg in args: - if arg == "--": - break - if arg in {"--cd", "-C"}: - value = next(args, None) - if value is not None: - directory = Path(value).expanduser() - elif arg.startswith("--cd="): - directory = Path(arg.partition("=")[2]).expanduser() - elif arg.startswith("-C"): - directory = Path(arg[2:].removeprefix("=")).expanduser() - return directory.resolve() - - class ModelVisibility(StrEnum): """A model's visibility in Codex's picker/APIs (mirrors Codex's ModelVisibility).""" diff --git a/src/ucode/smart_routing/orchestrator.py b/src/ucode/smart_routing/orchestrator.py index a538fbee0..2ed00d2ab 100644 --- a/src/ucode/smart_routing/orchestrator.py +++ b/src/ucode/smart_routing/orchestrator.py @@ -12,8 +12,7 @@ from collections.abc import Mapping from pathlib import Path -from ucode import codex_config, skills -from ucode.config_io import read_json_safe, read_toml_safe +from ucode import skills from ucode.smart_routing.hooks import sync_managed_hooks from ucode.smart_routing.session_env import effective_environment, session_env_path @@ -28,58 +27,6 @@ ) -def _legacy_plugin_ids(plugins: object) -> set[str]: - if not isinstance(plugins, dict): - return set() - return { - name - for name in plugins - if isinstance(name, str) and name.partition("@")[0] == "model-orchestrator" - } - - -def suppress_legacy_claude_plugin(settings: dict, user_settings_path: Path) -> bool: - """Suppress the installed predecessor for this launch, including when routing is off.""" - config_dir = Path(os.environ.get("CLAUDE_CONFIG_DIR", user_settings_path.parent)).expanduser() - installed = read_json_safe(config_dir / "plugins" / "installed_plugins.json") - user_settings = read_json_safe(config_dir / user_settings_path.name) - names = ( - _legacy_plugin_ids(installed.get("plugins")) - | _legacy_plugin_ids(user_settings.get("enabledPlugins")) - | _legacy_plugin_ids(settings.get("enabledPlugins")) - ) - if not names: - return False - plugins = settings.setdefault("enabledPlugins", {}) - if not isinstance(plugins, dict): - raise RuntimeError("Claude settings 'enabledPlugins' must be an object.") - plugins.update(dict.fromkeys(sorted(names), False)) - return True - - -def legacy_codex_plugin_config( - profile_path: Path | None = None, *, cwd: Path | None = None -) -> dict: - """Return a launch override; Isaac owns and may regenerate the saved plugin entries.""" - names: set[str] = set() - directory = (cwd or Path.cwd()).resolve() - paths = codex_config.codex_config_precedence_paths( - codex_config.codex_managed_config_path(), - profile_path or codex_config.DEFAULT_CODEX_CONFIG_PATH, - ) - # Only collect IDs to disable; never promote commands from untrusted project files. - project_paths = [] - for parent in (directory, *directory.parents): - project_paths.append(parent / ".codex" / "config.toml") - if (parent / ".git").exists(): - break - for path in (*paths, *project_paths): - names.update(_legacy_plugin_ids(read_toml_safe(path).get("plugins"))) - # Codex merges this table with the saved map. A dotted key cannot safely encode - # arbitrary marketplace names, and quoted dotted segments are treated literally. - return {"plugins": {name: {"enabled": False} for name in sorted(names)}} if names else {} - - def enabled(env: Mapping[str, str] | None = None) -> bool: from ucode.smart_routing.v2 import smart_routing_enabled diff --git a/src/ucode/smart_routing/v2.py b/src/ucode/smart_routing/v2.py index ba5515fa6..110645c1a 100644 --- a/src/ucode/smart_routing/v2.py +++ b/src/ucode/smart_routing/v2.py @@ -17,7 +17,6 @@ from ucode import config_io from ucode.codex_config import ( codex_config_args, - codex_working_directory, custom_catalog_models, custom_catalog_path, ) @@ -657,10 +656,6 @@ def launch_codex( overlay["model_catalog_json"] = str(catalog_path) overlay["hooks"] = _v2_hooks(state, available_models) overlay["features.hooks"] = True - legacy_plugin_config = orchestrator.legacy_codex_plugin_config( - cwd=codex_working_directory(tool_args) - ) - overlay.update(legacy_plugin_config) _prepare_smart_router_session("codex") # Codex constructs tool subprocess environments through its shell policy. # The skill's gate needs the same launch baseline as the routing hook, even @@ -712,16 +707,7 @@ def launch_codex( } ) tui = subprocess_cross_os.popen( - [ - binary, - *provider_args, - *codex_config_args(legacy_plugin_config), - "--remote", - tui_url, - "--model", - start_model, - *tool_args, - ] + [binary, *provider_args, "--remote", tui_url, "--model", start_model, *tool_args] ) try: returncode = tui.wait() diff --git a/tests/README.md b/tests/README.md index faf666608..c578cd3e4 100644 --- a/tests/README.md +++ b/tests/README.md @@ -143,10 +143,9 @@ including collapsed-output records, and excludes user echoes and assistant claim Dedicated regression coverage is missing for root-only orchestrator activation, compaction, retained skills in ineligible sessions, nested-session eligibility, -role-contract preservation, and isolation from legacy preference files. Legacy-plugin -suppression and preservation of saved settings and unrelated plugins/hooks are also -not covered by these journeys. Codex's native hook merging and project trust, -automatic delegation, and legacy-hook execution remain outside the integration suite. +role-contract preservation, and isolation from legacy preference files. Codex's +native hook merging and project trust, automatic delegation, and execution of +pre-existing hooks remain outside the integration suite. The portable Windows routing test checks native executable forwarding, generated hooks/plugins, caller arguments, and cleanup without Unix imports. It does not diff --git a/tests/conftest.py b/tests/conftest.py index 1ac6d2077..2b0c9f090 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -29,14 +29,12 @@ def _isolate_ucode_state(tmp_path, monkeypatch): it can never touch the developer's real ~/.ucode/state.json or invoke the privileged writer for an OS-managed agent config. """ - import ucode.codex_config as codex_config_mod import ucode.config_io as config_io_mod import ucode.databricks as databricks_mod import ucode.managed_config as managed_config_mod import ucode.managed_files as managed_files_mod import ucode.os_compatibility.subprocess_cross_os as subprocess_cross_os_mod import ucode.state as state_mod - from ucode.agents import claude as claude_mod from ucode.agents import codex as codex_mod state_dir = tmp_path / ".ucode" @@ -58,12 +56,6 @@ def _isolate_ucode_state(tmp_path, monkeypatch): codex_mod, "CODEX_MODEL_CATALOG_PATH", state_dir / "codex-model-catalog.json" ) monkeypatch.setattr(codex_mod, "CODEX_CONFIG_PATH", tmp_path / ".codex" / "ucode.config.toml") - # Launch-time plugin discovery must not read the developer's installed plugins. - monkeypatch.setattr(codex_config_mod, "DEFAULT_CODEX_CONFIG_PATH", codex_mod.CODEX_CONFIG_PATH) - monkeypatch.setattr(codex_config_mod, "codex_managed_config_path", lambda: None) - monkeypatch.setattr(claude_mod, "CLAUDE_USER_SETTINGS_PATH", tmp_path / ".claude/settings.json") - monkeypatch.delenv("CLAUDE_CONFIG_DIR", raising=False) - monkeypatch.delenv("CODEX_HOME", raising=False) def reject_privileged_write(path, _desired_text): pytest.fail( diff --git a/tests/integration/README.md b/tests/integration/README.md index fbdff561b..5eb6285dc 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -290,10 +290,8 @@ Their off-phase child is an explicit user-requested delegation; these journeys do not establish automatic orchestration behavior. Root-only activation, compaction, retained skills in ineligible sessions, nested-session eligibility, role-contract preservation, and isolation from legacy preference files -lack dedicated regression coverage. The journeys do not cover per-launch legacy-plugin -suppression or preservation of saved settings and unrelated plugins/hooks. -Codex's native hook merging, project trust, and execution of pre-existing hooks -are not exercised by this integration suite. +lack dedicated regression coverage. Codex's native hook merging, project trust, +and execution of pre-existing hooks are not exercised by this integration suite. The Claude on/off/on journey handles the visible permission prompt for the exact read-only orchestrator check. It verifies the command in the current dialog and diff --git a/tests/test_agent_claude.py b/tests/test_agent_claude.py index 94a8bcf63..e9bb3436a 100644 --- a/tests/test_agent_claude.py +++ b/tests/test_agent_claude.py @@ -2597,10 +2597,7 @@ def test_windows_launch_preserves_prompt_as_literal_argv(self, monkeypatch, tmp_ native_binary = tmp_path / "Claude Code" / "claude.exe" prompt = 'keep "quotes" & pipes | and %PATH% literal' calls: list[list[str]] = [] - # Keep the platform simulation local so filesystem readers use the real host paths. - monkeypatch.setattr( - claude, "os", SimpleNamespace(name="nt", path=os.path, environ=os.environ) - ) + monkeypatch.setattr(claude.os, "name", "nt") monkeypatch.setattr(claude.shutil, "which", lambda _binary: str(native_binary)) monkeypatch.setattr(claude, "exec_or_spawn", lambda argv: calls.append(argv)) diff --git a/tests/test_managed_files.py b/tests/test_managed_files.py index 6034a53ea..606a7ba8f 100644 --- a/tests/test_managed_files.py +++ b/tests/test_managed_files.py @@ -17,7 +17,6 @@ import ucode.codex_config as codex_config import ucode.config_io as config_io from ucode import managed_files -from ucode.codex_config import codex_managed_config_path _REAL_SUDO_REPLACE = managed_files._sudo_replace @@ -507,8 +506,7 @@ def test_real_managed_path_helpers_are_allowlisted(self, os_enum, monkeypatch): monkeypatch.setattr(codex_config, "current_os", lambda: os_enum) allowed = managed_files._SUDO_REPLACE_TARGETS[os_enum] assert claude_agent._managed_settings_path() in allowed - # Exercise the real helper, captured before launch fixtures isolate machine config. - assert codex_managed_config_path() in allowed + assert codex_config.codex_managed_config_path() in allowed @pytest.mark.skipif(sys.platform == "win32", reason="The managed writer is Unix-only") From dd27761ee12e34c7e71080ae7f56ed73bcdec7e0 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 18:12:23 +0000 Subject: [PATCH 10/25] Keep smart-routing toggle output concise --- src/ucode/cli.py | 6 ------ 1 file changed, 6 deletions(-) diff --git a/src/ucode/cli.py b/src/ucode/cli.py index 20a85c8f5..d2acd26bb 100644 --- a/src/ucode/cli.py +++ b/src/ucode/cli.py @@ -2311,12 +2311,6 @@ def _toggle_current_smart_routing_session(enabled: bool | None) -> bool: print_err(str(exc)) raise typer.Exit(1) from None print_success(f"Smart Router is {'on' if enabled else 'off'} for this session") - if enabled: - print_note("Automatic orchestration is on; apply the orchestrate skill to further work.") - else: - from ucode.smart_routing.orchestrator import DISABLED_CONTEXT - - print_note(DISABLED_CONTEXT) return True From b836dbe54277fb7c6bd744b3cc31a5be73bc4abd Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 18:12:51 +0000 Subject: [PATCH 11/25] Explain the bundled orchestrator skill --- src/ucode/skills.py | 1 + 1 file changed, 1 insertion(+) diff --git a/src/ucode/skills.py b/src/ucode/skills.py index d8c24cc65..26318c270 100644 --- a/src/ucode/skills.py +++ b/src/ucode/skills.py @@ -12,6 +12,7 @@ _LEGACY_SKILL_ROOTS = (".agents/skills",) _SKILL_NAME_PATTERN = re.compile(r"[a-z0-9]+(?:-[a-z0-9]+)*") SMART_ROUTER_SKILL = "smart-router" +# Install the delegation workflow in the same eligible sessions as Smart Router. ORCHESTRATOR_SKILL = "orchestrate" From 05b2ed5712186a1c04ad1b252e20805612ca7213 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 18:20:14 +0000 Subject: [PATCH 12/25] Gate orchestration behind an opt-in feature flag --- AGENTS.md | 4 +- README.md | 33 +++++++----- skills/orchestrate/README.md | 18 ++++--- skills/orchestrate/SKILL.md | 21 ++++---- skills/smart-router/SKILL.md | 10 ++-- src/ucode/constants.py | 1 + src/ucode/smart_routing/orchestrator.py | 50 +++++++++++-------- src/ucode/smart_routing/v2.py | 10 +++- tests/README.md | 10 ++-- tests/integration/README.md | 8 +-- .../test_ug_smart_routing_hooks.py | 39 +++++++++++---- tests/test_claude_smart_routing_v2.py | 4 +- tests/test_claude_windows_smart_routing.py | 8 ++- 13 files changed, 136 insertions(+), 80 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index bc03fddd7..48b9b4053 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -70,7 +70,7 @@ Fields live in `~/.claude/ucode-settings.json` and the OS-managed settings file | Tracing | Ignore | Create/replace | The seven `CLAUDE_CODE_*`/`OTEL_*` trace keys and `otelHeadersHelper`; only when the config enables tracing | | `managedMcpServers` | Ignore | Merge | Add/update the config's MCP server entries; other entries left alone | | Smart-routing hooks | Merge | Merge | `PreToolUse`, `SessionStart`, `SubagentStart`; only `ug`'s own marked handlers, other hooks left alone | -| Orchestrator hooks | Merge | Merge | Launch-only `UserPromptSubmit` and compact `SessionStart` handlers for smart-routed sessions; read the same session controls as routing | +| Orchestrator hooks | Merge | Merge | Launch-only `UserPromptSubmit` and compact `SessionStart` handlers for smart-routed sessions with `ENABLE_ORCHESTRATION=1`; read the same session controls as routing | @@ -87,6 +87,6 @@ Fields live in `~/.codex/ucode.config.toml` and `/etc/codex/managed_config.toml` | `http_headers` | Merge | Merge | In `[model_providers.Databricks]`; merge `ug`'s routing headers by name, admin headers added under managed config | | `model_catalog_json` | Create/replace | Create/replace | In `~/.codex/config.toml`; `ug`'s own catalog reference, for a static model list | | `mcp_servers` | Ignore | Merge | Managed file; add/update the config's MCP server entries, other entries left alone | -| Smart-routing and orchestrator hooks | Merge | Merge | Launch-only `PreToolUse`, `UserPromptSubmit`, and `SessionStart` handlers; Codex combines them with its native hook sources and applies project trust; `features.hooks` is enabled for a smart-routed launch | +| Smart-routing and orchestrator hooks | Merge | Merge | Launch-only `PreToolUse`, plus `UserPromptSubmit` and compact `SessionStart` with `ENABLE_ORCHESTRATION=1`; Codex combines them with its native hook sources and applies project trust; `features.hooks` is enabled for a smart-routed launch | diff --git a/README.md b/README.md index 8573808b4..575abfd64 100644 --- a/README.md +++ b/README.md @@ -248,19 +248,26 @@ The generated shell hooks expect Git Bash; PowerShell-only setups are not covere ### Automatic orchestration -Smart-routed Claude and Codex sessions install the bundled `orchestrate` and -`smart-router` skills. The orchestrator assigns bounded work to explorer, -researcher, worker, tester, and reviewer roles while the root plans, integrates, -and verifies results. Easy tasks and explicit requests not to delegate stay in -the root. - -Orchestration follows the existing smart-routing launch eligibility and session -controls; it has no separate rollout flag. Turning Smart Router off stops new -automatic delegation. Turning it on -restores orchestration. Explicit user requests for subagents still use normal -harness behavior while routing is off. Stored skill files do not activate it in -later non-routed sessions. The existing Isaac pilot gate and UG launch exclusions -are unchanged. +Smart-routed Claude and Codex sessions install `smart-router`. Set +`ENABLE_ORCHESTRATION=1` at launch to also install and activate the bundled +`orchestrate` skill; orchestration is off by default. For example: + +```bash +ENABLE_ORCHESTRATION=1 ENABLE_SMART_ROUTING_SUBAGENT_ONLY=1 ug claude +``` + +Use `ug codex` in the same command for Codex. The orchestrator assigns bounded work +to explorer, researcher, worker, tester, and reviewer roles while the root plans, +integrates, and verifies results. Easy tasks and explicit requests not to delegate +stay in the root. + +Once opted in, orchestration follows the existing smart-routing launch eligibility +and session controls. Turning Smart Router off stops new automatic delegation; +turning it on restores orchestration only in opted-in sessions. Explicit user +requests for subagents still use normal harness behavior while routing is off. +Stored skill files do not activate orchestration when the feature flag is unset or +`ENABLE_ORCHESTRATION=0`, or in non-routed sessions. Existing Isaac pilot gating +and UG launch exclusions still apply. UG supplies its own hooks; Codex combines them with existing hooks and applies project trust. Smart routing selects subagent models; separate role-model diff --git a/skills/orchestrate/README.md b/skills/orchestrate/README.md index 7e962944c..f196637dc 100644 --- a/skills/orchestrate/README.md +++ b/skills/orchestrate/README.md @@ -1,14 +1,16 @@ # UG model orchestrator UG bundles the `orchestrate` workflow and five Claude role definitions from -`model-orchestrator` 0.4.10. Smart-routed -Claude and Codex launches install this skill alongside `smart-router`. +`model-orchestrator` 0.4.10. Smart-routed Claude and Codex launches install and +activate this skill alongside `smart-router` only with `ENABLE_ORCHESTRATION=1`. +The feature is off by default; routing alone installs only `smart-router`. The workflow is injected before root prompts and after compaction. Its activation -and pre-delegation checks require a UG smart-routing session and read the same -session controls as the routing hooks. Turning Smart Router off stops new -automatic delegation and supersedes the previous workflow. Turning it on restores -both features. An installed skill or saved model preference cannot enable them. +and pre-delegation checks require the feature flag and a UG smart-routing session, +and read the same session controls as the routing hooks. Turning Smart Router off +stops new automatic delegation and supersedes the previous workflow. Turning it on restores +orchestration only in opted-in sessions. An installed skill or saved model +preference cannot enable it. Explicit user requests for subagents still use native harness behavior while routing is off, without the orchestrator's workflow or routing check. User instructions take precedence, and easy tasks remain in the root. @@ -31,7 +33,9 @@ model selection and requires no preference setup, locking, or recovery. The [skill](SKILL.md) checks eligibility with `"$UCODE_SMART_ROUTER_PYTHON" -m ucode.smart_routing.orchestrator --check` before delegating. This command reads session controls without reading or writing -preference files, succeeds silently when enabled, and exits nonzero when disabled. +preference files, succeeds silently when both orchestration and routing are enabled, +and exits nonzero otherwise. Retained skill files stay inactive when the feature +flag is unset or `0`. ## Attribution diff --git a/skills/orchestrate/SKILL.md b/skills/orchestrate/SKILL.md index fd0a51de8..8c49d51e7 100644 --- a/skills/orchestrate/SKILL.md +++ b/skills/orchestrate/SKILL.md @@ -1,6 +1,6 @@ --- name: orchestrate -description: Coordinate substantive development with native subagents while Unity Gateway smart routing is enabled. Follow the routing-state check before using this workflow. Skip easy tasks and explicit no-subagent requests. +description: Coordinate substantive development with native subagents when ENABLE_ORCHESTRATION=1 and Unity Gateway smart routing is enabled. Follow the routing-state check before using this workflow. Skip easy tasks and explicit no-subagent requests. model: inherit argument-hint: "[task]" metadata: @@ -11,21 +11,23 @@ metadata: ## Smart-routing gate -This workflow is active only in a UG-launched smart-routing session while routing -is enabled. Installed skill files and old context do not enable it. Before -**every new delegation under this workflow**, check the same session controls +This workflow is active only in a UG-launched smart-routing session with +`ENABLE_ORCHESTRATION=1` and routing enabled. Installed skill files and old +context do not enable it. Before **every new delegation under this workflow**, +check the same session controls as the routing hooks with the launching interpreter: ```text "$UCODE_SMART_ROUTER_PYTHON" -m ucode.smart_routing.orchestrator --check ``` -The command succeeds silently when routing is enabled. In PowerShell, use -`& $env:UCODE_SMART_ROUTER_PYTHON` in place of `"$UCODE_SMART_ROUTER_PYTHON"`. +The command succeeds silently when both orchestration and routing are enabled. +In PowerShell, use `& $env:UCODE_SMART_ROUTER_PYTHON` in place of +`"$UCODE_SMART_ROUTER_PYTHON"`. If the interpreter is absent or the command fails, do not use this workflow; report the problem and continue authorized work in the root. Never choose -another Python from PATH, set routing flags, or create a session to bypass -this check. +another Python from PATH, set orchestration or routing flags, or create a session +to bypass this check. Turning Smart Router off also turns this workflow off immediately and supersedes earlier orchestration instructions. Do not start new automatic delegation or use @@ -33,7 +35,8 @@ orchestrator role models as a fallback. Continue in the root unless the user explicitly requests a subagent; honor that request using the native tool and normal harness model selection, without this workflow or its routing check. Keep routing off and collect results from existing children. Turning Smart Router back on restores -this workflow. Use the `smart-router` skill only when the user asks to change routing. +this workflow only if the session was launched with `ENABLE_ORCHESTRATION=1`. +Use the `smart-router` skill only when the user asks to change routing. ## Workflow diff --git a/skills/smart-router/SKILL.md b/skills/smart-router/SKILL.md index 07118d7b7..defa78046 100644 --- a/skills/smart-router/SKILL.md +++ b/skills/smart-router/SKILL.md @@ -1,6 +1,6 @@ --- name: smart-router -description: Enable or disable Unity Gateway subagent model routing and automatic orchestration together for the current smart-routed Claude or Codex session. +description: Enable or disable Unity Gateway subagent model routing for the current smart-routed Claude or Codex session, along with automatic orchestration when opted in. allowed-tools: Bash("$UCODE_SMART_ROUTER_PYTHON" -m ucode.cli claude --enable-smart-routing), Bash("$UCODE_SMART_ROUTER_PYTHON" -m ucode.cli claude --disable-smart-routing), Bash("$UCODE_SMART_ROUTER_PYTHON" -m ucode.cli codex --enable-smart-routing), Bash("$UCODE_SMART_ROUTER_PYTHON" -m ucode.cli codex --disable-smart-routing) metadata: version: "1.2.0" @@ -22,10 +22,12 @@ If `UCODE_SMART_ROUTER_PYTHON` or `UCODE_SESSION_ENV_FILE` is unset, ask the use to restart through an updated Unity Gateway with smart routing enabled. With no argument, explain that only `on` and `off` are accepted. Do not edit the state file. -This affects subsequent subagent model selection and automatic orchestration in the current -session, not the root model or first prompt. When turned off, earlier orchestrate instructions +This affects subsequent subagent model selection in the current session, not the root model +or first prompt. In sessions launched with `ENABLE_ORCHESTRATION=1`, it also controls automatic +orchestration. When turned off, earlier orchestrate instructions are superseded: do not start new automatic delegation or fall back to orchestrator role models. Continue in the root unless the user explicitly requests a subagent. Honor that request using native tools and normal harness model selection, without the orchestrator or its resolution helper; keep routing off. Existing children can finish. When turned on, apply the orchestrate -skill to further work. Return the command's result. +skill to further work only if the session was launched with `ENABLE_ORCHESTRATION=1`. +Do not change the orchestration feature flag. Return the command's result. diff --git a/src/ucode/constants.py b/src/ucode/constants.py index 1db9930e6..25b847964 100644 --- a/src/ucode/constants.py +++ b/src/ucode/constants.py @@ -5,6 +5,7 @@ ENABLE_SMART_ROUTING_ENV_VAR = "ENABLE_SMART_ROUTING_V2" ENABLE_SUBAGENT_ROUTING_ENV_VAR = "ENABLE_SMART_ROUTING_SUBAGENT_ONLY" +ENABLE_ORCHESTRATION_ENV_VAR = "ENABLE_ORCHESTRATION" SMART_ROUTING_ENV_KEYS = ( ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, diff --git a/src/ucode/smart_routing/orchestrator.py b/src/ucode/smart_routing/orchestrator.py index 2ed00d2ab..e79a8410d 100644 --- a/src/ucode/smart_routing/orchestrator.py +++ b/src/ucode/smart_routing/orchestrator.py @@ -1,4 +1,4 @@ -"""Activate the bundled orchestrator only in an enabled smart-routing session.""" +"""Activate the bundled orchestrator in opted-in smart-routing sessions.""" from __future__ import annotations @@ -13,24 +13,32 @@ from pathlib import Path from ucode import skills +from ucode.constants import ENABLE_ORCHESTRATION_ENV_VAR from ucode.smart_routing.hooks import sync_managed_hooks from ucode.smart_routing.session_env import effective_environment, session_env_path HOOK_MODULE = "ucode.smart_routing.orchestrator" DISABLED_CONTEXT = ( - "UG automatic orchestration is off because smart routing is off for this session. " + "UG automatic orchestration is off for this session. " "This supersedes any earlier model-orchestrator workflow: do not start new automatic " "delegation or fall back to orchestrator role models. Explicit user requests for subagents " - "still use native tools and normal harness model selection, without the orchestrator " - "or its routing check; keep routing off. Otherwise continue the task in the root. " + "still use native tools and the current Smart Router setting, without the orchestrator " + "or its routing check. Otherwise continue the task in the root. " "Collect results from children already running." ) +def feature_enabled(env: Mapping[str, str] | None = None) -> bool: + source = os.environ if env is None else env + return source.get(ENABLE_ORCHESTRATION_ENV_VAR) == "1" + + def enabled(env: Mapping[str, str] | None = None) -> bool: from ucode.smart_routing.v2 import smart_routing_enabled source = os.environ if env is None else env + if not feature_enabled(source): + return False if source.get("ISAAC_LAUNCH_MODE", "").strip().lower() == "omni": return False try: @@ -53,26 +61,26 @@ def skill_directory() -> Path: def add_claude_agents(plugin_dir: Path) -> None: """Load roles alongside the router's exact-model agents, only for this launch.""" - shutil.copytree(skill_directory() / "agents", plugin_dir / "agents", dirs_exist_ok=True) + if feature_enabled(): + shutil.copytree(skill_directory() / "agents", plugin_dir / "agents", dirs_exist_ok=True) def sync_hooks(doc: dict, *, agent: str) -> None: - argv = [sys.executable, "-m", HOOK_MODULE] - hook = { - "type": "command", - "command": shlex.join(argv), - "timeout": 5, - } - if agent == "codex": - hook["command_windows"] = subprocess.list2cmdline(argv) - sync_managed_hooks( - doc, - HOOK_MODULE, - { + groups = {} + if feature_enabled(): + argv = [sys.executable, "-m", HOOK_MODULE] + hook = { + "type": "command", + "command": shlex.join(argv), + "timeout": 5, + } + if agent == "codex": + hook["command_windows"] = subprocess.list2cmdline(argv) + groups = { "UserPromptSubmit": [{"hooks": [hook]}], "SessionStart": [{"matcher": "compact", "hooks": [hook]}], - }, - ) + } + sync_managed_hooks(doc, HOOK_MODULE, groups) def hook_output(payload: object) -> dict | None: @@ -92,7 +100,7 @@ def hook_output(payload: object) -> dict | None: return None context = ( "Apply the UG model-orchestrator workflow to this task. " - "Smart routing and automatic orchestration share the same session controls.\n" + "Orchestration is opted in and follows Smart Router's session controls.\n" f"Skill directory: {directory}\n\n{workflow}" ) return {"hookSpecificOutput": {"hookEventName": event, "additionalContext": context}} @@ -103,7 +111,7 @@ def main(argv: list[str] | None = None) -> None: parser.add_argument( "--check", action="store_true", - help="Exit successfully only in an enabled smart-routing session.", + help="Exit successfully only in an opted-in, enabled smart-routing session.", ) if parser.parse_args(argv).check: try: diff --git a/src/ucode/smart_routing/v2.py b/src/ucode/smart_routing/v2.py index 110645c1a..467f1fa6c 100644 --- a/src/ucode/smart_routing/v2.py +++ b/src/ucode/smart_routing/v2.py @@ -27,6 +27,7 @@ write_text_file, ) from ucode.constants import ( + ENABLE_ORCHESTRATION_ENV_VAR, ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, LOOPBACK_HOST, @@ -83,7 +84,10 @@ class ClaudeRoutingSetupError(RuntimeError): def _prepare_smart_router_session(agent: str) -> Path: - for skill in (SMART_ROUTER_SKILL, ORCHESTRATOR_SKILL): + skills = [SMART_ROUTER_SKILL] + if orchestrator.feature_enabled(): + skills.append(ORCHESTRATOR_SKILL) + for skill in skills: try: install_skill(skill, agent, config_io.APP_DIR.parent) except (OSError, RuntimeError) as exc: @@ -524,6 +528,7 @@ def launch_claude( if not isinstance(env, dict): raise RuntimeError("Claude settings 'env' must be an object for smart routing.") env.pop("CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY", None) + env[ENABLE_ORCHESTRATION_ENV_VAR] = "1" if orchestrator.feature_enabled() else "0" if route_first_prompt: env[ENABLE_SMART_ROUTING_ENV_VAR] = "1" else: @@ -663,6 +668,9 @@ def launch_codex( for key in (SESSION_ENV_VAR, SESSION_PYTHON_ENV_VAR, *SMART_ROUTING_ENV_KEYS): if key in os.environ: overlay[f"shell_environment_policy.set.{key}"] = os.environ[key] + overlay[f"shell_environment_policy.set.{ENABLE_ORCHESTRATION_ENV_VAR}"] = ( + "1" if orchestrator.feature_enabled() else "0" + ) config_args = codex_config_args(overlay) if not first_prompt_routing_enabled(): # Subagent-only routing needs neither the app-server nor the interposer: diff --git a/tests/README.md b/tests/README.md index c578cd3e4..ee3fe8463 100644 --- a/tests/README.md +++ b/tests/README.md @@ -131,9 +131,11 @@ that Claude settings and Codex's shell policy carry the interpreter and session These are component checks; they do not establish native skill permission matching or PowerShell execution. -The toggle integration journeys require both bundled skills, verify the saved -session controls and native tool-result confirmation after each toggle, and -explicitly request their children, including while routing is off. Claude's journey +The toggle integration journeys run with `ENABLE_ORCHESTRATION` unset and with +`ENABLE_ORCHESTRATION=1`. They require only `smart-router` by default and both +bundled skills when opted in, verify the saved session controls and native +tool-result confirmation after each toggle, and explicitly request their children, +including while routing is off. Claude's journey answers the visible permission prompt for the exact read-only orchestrator check, using the dialog's command because pending calls may not yet be in the transcript. Evidence tests reject added shell commands, commands in scrollback or descriptions, @@ -196,7 +198,7 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_ug_codex_app_reports_unknown_argument` | Pass an invalid option directly to `ug codex app`, routing off/on | Real Codex parser error and status preserved | | `test_ug_codex_app_server_client_initializes` | Connect a stdio client, direct/`--` separator, routing off/on | Actual JSON-RPC initialize response; no non-JSON stdout; no routing | | `test_smart_routing_claude_route_subagent_hook`, `test_smart_routing_codex_route_subagent_hook` | Pipe a real PreToolUse spawn payload to the installed route-subagent hook with subagent-only routing enabled | Allow decision against the live router; requested model replaced by a routed agent definition (Claude) or bundled catalog slug (Codex) from the offered models; one audited decision matching the session and task | -| `test_smart_router_skill_toggles_claude_subagent_routing`, `test_smart_router_skill_toggles_codex_subagent_routing` | Configure, launch a real subagent-only TUI, then spawn tagged children while invoking the installed Smart Router skill to switch routing on -> off -> on in the same session | All three native children complete; only enabled phases show the subagent banner and produce a live routing decision correlated with the child; no first-prompt routing wrapper; normal exit | +| `test_smart_router_skill_toggles_claude_subagent_routing`, `test_smart_router_skill_toggles_codex_subagent_routing` | Configure, launch a real subagent-only TUI with orchestration unset or opted in, then spawn tagged children while invoking the installed Smart Router skill to switch routing on -> off -> on in the same session | Only `smart-router` is installed by default; opt-in also installs `orchestrate`; all three native children complete; only routing-enabled phases show the subagent banner and produce a live routing decision correlated with the child; no first-prompt routing wrapper; normal exit | | `test_ug_configure_claude_repeat_and_revert`, `test_ug_configure_codex_repeat_and_revert` | Configure twice over user settings; complete a task; revert twice | Settings preserved; no bearer in ug state; generated config removed; status unconfigured | | `test_ug_configure_claude_cleans_stale_skills_mcp_on_workspace_switch` | Configure the first workspace, register its skills MCP, switch to a second real workspace, and use Claude | Old registration removed from Claude and the new workspace state; old workspace bucket preserved; repeat configure stays clean; real file task completes on the second workspace | | `test_ug_configure_claude_rejects_invalid_credentials`, `test_ug_configure_codex_rejects_invalid_credentials` | Configure with a rejected bearer against the real workspace | Authentication failure; no successful saved setup | diff --git a/tests/integration/README.md b/tests/integration/README.md index 5eb6285dc..92b40cdde 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -281,9 +281,11 @@ PATH conflicts for the Smart Router skill have subprocess/component coverage in `ug` first in PATH. The live journeys above do not inject a second installation or establish PowerShell command execution. -The toggle journeys require both bundled skills (`orchestrate` and `smart-router`) -to be installed. They verify the saved session controls, a new CLI confirmation in -the native tool-result records, and a new assistant answer after each skill invocation. +The toggle journeys run with `ENABLE_ORCHESTRATION` unset and with +`ENABLE_ORCHESTRATION=1`. They require only `smart-router` by default and both +`orchestrate` and `smart-router` when opted in. They verify the saved session +controls, a new CLI confirmation in the native tool-result records, and a new +assistant answer after each skill invocation. Collapsed terminal output is allowed; the answer need not repeat the CLI's exact wording. Each following child still verifies whether a routing decision occurred. Their off-phase child is an explicit user-requested delegation; diff --git a/tests/integration/test_ug_smart_routing_hooks.py b/tests/integration/test_ug_smart_routing_hooks.py index 43eeae741..187fb26b0 100644 --- a/tests/integration/test_ug_smart_routing_hooks.py +++ b/tests/integration/test_ug_smart_routing_hooks.py @@ -147,7 +147,12 @@ def _toggle_with_skill(tui, session, agent: str, enabled: bool) -> None: for path in skill_root.iterdir() if path.is_dir() and path.name not in ignored_skills ) - assert installed_skills == ["orchestrate", "smart-router"], installed_skills + expected_skills = ( + ["orchestrate", "smart-router"] + if session.env.get("ENABLE_ORCHESTRATION") == "1" + else ["smart-router"] + ) + assert installed_skills == expected_skills, installed_skills state = "on" if enabled else "off" invocation = f"/smart-router {state}" if agent == "claude" else f"$smart-router {state}" @@ -301,13 +306,19 @@ def test_smart_routing_codex_route_subagent_hook(live_session, workspace): @pytest.mark.live @pytest.mark.claude @pytest.mark.managed_fixture -def test_smart_router_skill_toggles_claude_subagent_routing(live_session, workspace, tmp_path): - """Scenario: launch Claude with subagent routing enabled, spawn a child, invoke the +@pytest.mark.parametrize( + "orchestration_enabled", [False, True], ids=["routing-only", "orchestration"] +) +def test_smart_router_skill_toggles_claude_subagent_routing( + live_session, workspace, tmp_path, orchestration_enabled +): + """Scenario: launch Claude with subagent routing enabled and orchestration unset + or opted in through ENABLE_ORCHESTRATION=1, spawn a child, invoke the installed Smart Router skill to turn routing off, spawn another child, turn routing back on through the skill, and spawn a third child in the same real TUI session. - Expected: Smart Router and orchestrate are the only user-installed Claude skills; - each invocation records the CLI confirmation in the native transcript and changes the saved + Expected: only Smart Router is installed by default; opting in also installs orchestrate. + Each invocation records the CLI confirmation in the native transcript and changes the saved routing controls, even with collapsed terminal output; all three uniquely tagged calculations complete in native child sessions; only the first and third show the subagent-routing banner and produce live gateway decisions correlated with those children. @@ -318,6 +329,8 @@ def test_smart_router_skill_toggles_claude_subagent_routing(live_session, worksp session.env["TMPDIR"] = str(tmp_path) session.env["ENABLE_SMART_ROUTING_V2"] = "1" session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" + if orchestration_enabled: + session.env["ENABLE_ORCHESTRATION"] = "1" config = build_coding_agent_config( "CODING_AGENT_CLAUDE_CODE", build_claude_agent_config(CLAUDE_SMART_ROUTING_MODELS, smart_routing=True), @@ -351,13 +364,19 @@ def test_smart_router_skill_toggles_claude_subagent_routing(live_session, worksp @pytest.mark.live @pytest.mark.codex @pytest.mark.managed_fixture -def test_smart_router_skill_toggles_codex_subagent_routing(live_session, workspace, tmp_path): - """Scenario: launch Codex with subagent routing enabled, spawn a child, invoke the +@pytest.mark.parametrize( + "orchestration_enabled", [False, True], ids=["routing-only", "orchestration"] +) +def test_smart_router_skill_toggles_codex_subagent_routing( + live_session, workspace, tmp_path, orchestration_enabled +): + """Scenario: launch Codex with subagent routing enabled and orchestration unset + or opted in through ENABLE_ORCHESTRATION=1, spawn a child, invoke the installed Smart Router skill to turn routing off, spawn another child, turn routing back on through the skill, and spawn a third child in the same real TUI session. - Expected: Smart Router and orchestrate are the only user-installed Codex skills; - each invocation records the CLI confirmation in the native transcript and changes the saved + Expected: only Smart Router is installed by default; opting in also installs orchestrate. + Each invocation records the CLI confirmation in the native transcript and changes the saved routing controls, even with collapsed terminal output; all three uniquely tagged calculations complete in native child sessions; only the first and third show the subagent-routing banner and produce live gateway decisions correlated with those children. @@ -367,6 +386,8 @@ def test_smart_router_skill_toggles_codex_subagent_routing(live_session, workspa session.env["TMPDIR"] = str(tmp_path) session.env["ENABLE_SMART_ROUTING_V2"] = "1" session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" + if orchestration_enabled: + session.env["ENABLE_ORCHESTRATION"] = "1" config = build_coding_agent_config( "CODING_AGENT_CODEX", build_codex_agent_config(models=CODEX_SMART_ROUTING_MODELS, smart_routing=True), diff --git a/tests/test_claude_smart_routing_v2.py b/tests/test_claude_smart_routing_v2.py index 1cffba3ff..6179e3a75 100644 --- a/tests/test_claude_smart_routing_v2.py +++ b/tests/test_claude_smart_routing_v2.py @@ -19,7 +19,7 @@ def _plugin_agent_models(plugin_dir: Path) -> set[str]: models = set() - for agent_path in (plugin_dir / "agents").glob(f"{v2.CLAUDE_ROUTED_AGENT_PREFIX}*.md"): + for agent_path in (plugin_dir / "agents").glob("*.md"): model_line = next( line for line in agent_path.read_text().splitlines() if line.startswith("model: ") ) @@ -482,7 +482,7 @@ def send_signal(self, _signal): assert v2.ENABLE_SMART_ROUTING_ENV_VAR not in env assert claude_hooks.FIRST_PROMPT_SOCKET_ENV not in env # Subagent routing is fully wired; only the first-prompt machinery is absent. - assert "route-first-prompt" not in str(settings["hooks"]) + assert "UserPromptSubmit" not in settings["hooks"] assert "route-subagent" in str(settings["hooks"]["PreToolUse"]) assert settings["modelOverrides"] == {"claude-opus-4-8": "system.ai.claude-opus-4-8"} assert captured["plugin_models"] == {"system.ai.claude-opus-4-8"} diff --git a/tests/test_claude_windows_smart_routing.py b/tests/test_claude_windows_smart_routing.py index 62741fae5..b6db0f462 100644 --- a/tests/test_claude_windows_smart_routing.py +++ b/tests/test_claude_windows_smart_routing.py @@ -10,12 +10,12 @@ from ucode.agents import claude from ucode.databricks import AnthropicModelCatalog -from ucode.smart_routing import orchestrator, session_env, v2 +from ucode.smart_routing import session_env, v2 def _plugin_agent_models(plugin_dir: Path) -> set[str]: models = set() - for agent_path in (plugin_dir / "agents").glob(f"{v2.CLAUDE_ROUTED_AGENT_PREFIX}*.md"): + for agent_path in (plugin_dir / "agents").glob("*.md"): model_line = next( line for line in agent_path.read_text().splitlines() if line.startswith("model: ") ) @@ -43,8 +43,6 @@ def test_windows_subagent_routing_uses_native_binary_without_unix_imports(tmp_pa warnings: list[str] = [] host_os_name = v2.os.name - skill_directory = orchestrator.skill_directory() - monkeypatch.setattr(orchestrator, "skill_directory", lambda: skill_directory) path_type = type(tmp_path) monkeypatch.setattr(v2.os, "name", "nt") # ``pathlib.Path`` follows the process-wide os.name even on this Linux test host. Keep the @@ -120,7 +118,7 @@ def send_signal(self, _signal): assert captured["plugin_models"] == {"system.ai.claude-opus-4-8"} assert settings["env"][v2.ENABLE_SUBAGENT_ROUTING_ENV_VAR] == "1" assert v2.ENABLE_SMART_ROUTING_ENV_VAR not in settings["env"] - assert "route-first-prompt" not in str(settings["hooks"]) + assert "UserPromptSubmit" not in settings["hooks"] assert "route-subagent" in str(settings["hooks"]["PreToolUse"]) assert settings["modelOverrides"] == {"claude-opus-4-8": "system.ai.claude-opus-4-8"} assert not captured["settings_path"].exists() From f41bec035e652fa4f5718a910e81c97b41d84337 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 18:31:37 +0000 Subject: [PATCH 13/25] Simplify orchestrator activation to automatic hooks --- README.md | 6 ++- skills/orchestrate/README.md | 25 +++++------ skills/orchestrate/SKILL.md | 36 ++++++---------- skills/smart-router/SKILL.md | 4 +- src/ucode/smart_routing/orchestrator.py | 32 +++----------- src/ucode/smart_routing/v2.py | 2 +- tests/README.md | 6 +-- tests/integration/README.md | 6 --- .../test_ug_smart_routing_hooks.py | 27 +----------- tests/integration/utils/evidence.py | 20 --------- tests/test_integration_evidence.py | 43 ------------------- 11 files changed, 39 insertions(+), 168 deletions(-) diff --git a/README.md b/README.md index 575abfd64..acbf4441b 100644 --- a/README.md +++ b/README.md @@ -262,13 +262,17 @@ integrates, and verifies results. Easy tasks and explicit requests not to delega stay in the root. Once opted in, orchestration follows the existing smart-routing launch eligibility -and session controls. Turning Smart Router off stops new automatic delegation; +and session controls. Turning Smart Router off through its skill stops new automatic delegation; turning it on restores orchestration only in opted-in sessions. Explicit user requests for subagents still use normal harness behavior while routing is off. Stored skill files do not activate orchestration when the feature flag is unset or `ENABLE_ORCHESTRATION=0`, or in non-routed sessions. Existing Isaac pilot gating and UG launch exclusions still apply. +Hooks refresh orchestration state before each prompt and after compaction. A +state change made outside the conversation is observed at the next hook; model +routing still checks the controls for each subagent. + UG supplies its own hooks; Codex combines them with existing hooks and applies project trust. Smart routing selects subagent models; separate role-model preferences are ignored and their files are left untouched. See the bundled diff --git a/skills/orchestrate/README.md b/skills/orchestrate/README.md index f196637dc..d39950757 100644 --- a/skills/orchestrate/README.md +++ b/skills/orchestrate/README.md @@ -5,14 +5,16 @@ UG bundles the `orchestrate` workflow and five Claude role definitions from activate this skill alongside `smart-router` only with `ENABLE_ORCHESTRATION=1`. The feature is off by default; routing alone installs only `smart-router`. -The workflow is injected before root prompts and after compaction. Its activation -and pre-delegation checks require the feature flag and a UG smart-routing session, -and read the same session controls as the routing hooks. Turning Smart Router off -stops new automatic delegation and supersedes the previous workflow. Turning it on restores -orchestration only in opted-in sessions. An installed skill or saved model -preference cannot enable it. +The workflow is injected before root prompts and after compaction. The hook checks +the feature flag, UG session, and current routing controls. The skill uses that +activation context without running a separate check before delegation. +Turning Smart Router off through its skill stops new automatic delegation and +supersedes the previous workflow. Turning it on restores orchestration only in +opted-in sessions. A change made outside the conversation is observed at the next +prompt or compaction; model routing still checks the controls for each subagent. +An installed skill or saved model preference cannot enable orchestration. Explicit user requests for subagents still use native harness behavior while routing -is off, without the orchestrator's workflow or routing check. +is off, without the orchestrator's workflow. User instructions take precedence, and easy tasks remain in the root. Claude loads the bundled roles as `ug-smart-router:` in its temporary @@ -22,7 +24,7 @@ Role instructions belong in each task prompt because routing may replace the requested Claude role or Codex model. Hook approval in the native `/hooks` UI is still required where the harness prompts for it. -## Routing check +## Model selection Separate role-model preferences are not used by the UG workflow. Existing `.model-orchestrator.json` project preferences, @@ -30,13 +32,6 @@ Separate role-model preferences are not used by the UG workflow. Existing custom Claude agents are left untouched. The bundled workflow uses the router's model selection and requires no preference setup, locking, or recovery. -The [skill](SKILL.md) checks eligibility with -`"$UCODE_SMART_ROUTER_PYTHON" -m ucode.smart_routing.orchestrator --check` before -delegating. This command reads session controls without reading or writing -preference files, succeeds silently when both orchestration and routing are enabled, -and exits nonzero otherwise. Retained skill files stay inactive when the feature -flag is unset or `0`. - ## Attribution Migrated from the Databricks `model-orchestrator` plugin 0.4.10 by Arnav Singhvi. diff --git a/skills/orchestrate/SKILL.md b/skills/orchestrate/SKILL.md index 8c49d51e7..99b2441bf 100644 --- a/skills/orchestrate/SKILL.md +++ b/skills/orchestrate/SKILL.md @@ -1,6 +1,6 @@ --- name: orchestrate -description: Coordinate substantive development with native subagents when ENABLE_ORCHESTRATION=1 and Unity Gateway smart routing is enabled. Follow the routing-state check before using this workflow. Skip easy tasks and explicit no-subagent requests. +description: Coordinate substantive development with native subagents when ENABLE_ORCHESTRATION=1 and Unity Gateway smart routing is enabled. Follow UG's activation context. Skip easy tasks and explicit no-subagent requests. model: inherit argument-hint: "[task]" metadata: @@ -9,31 +9,20 @@ metadata: # Model orchestrator -## Smart-routing gate +## Activation -This workflow is active only in a UG-launched smart-routing session with -`ENABLE_ORCHESTRATION=1` and routing enabled. Installed skill files and old -context do not enable it. Before **every new delegation under this workflow**, -check the same session controls -as the routing hooks with the launching interpreter: +UG's prompt and compaction hooks activate this workflow only in an eligible +smart-routing session launched with `ENABLE_ORCHESTRATION=1` while routing is on. +Follow the latest UG activation context and successful Smart Router toggles; +installed skill files and old context do not enable it. Use that context without +running a separate pre-delegation check. Do not set flags or create a session to +activate this workflow. -```text -"$UCODE_SMART_ROUTER_PYTHON" -m ucode.smart_routing.orchestrator --check -``` - -The command succeeds silently when both orchestration and routing are enabled. -In PowerShell, use `& $env:UCODE_SMART_ROUTER_PYTHON` in place of -`"$UCODE_SMART_ROUTER_PYTHON"`. -If the interpreter is absent or the command fails, do not use this workflow; -report the problem and continue authorized work in the root. Never choose -another Python from PATH, set orchestration or routing flags, or create a session -to bypass this check. - -Turning Smart Router off also turns this workflow off immediately and supersedes +Turning Smart Router off through its skill stops this workflow and supersedes earlier orchestration instructions. Do not start new automatic delegation or use orchestrator role models as a fallback. Continue in the root unless the user explicitly requests a subagent; honor that request using the native tool and normal harness -model selection, without this workflow or its routing check. Keep routing off +model selection, without this workflow. Keep routing off and collect results from existing children. Turning Smart Router back on restores this workflow only if the session was launched with `ENABLE_ORCHESTRATION=1`. Use the `smart-router` skill only when the user asks to change routing. @@ -48,9 +37,8 @@ Never change providers, credentials, permissions, sandbox, unrelated settings, or concurrency limits. Report conflicts with existing mandatory orchestration rules or model policies. -Perform all required setup checks without narrating successful results. Before -delegating, describe the task split in at most one short sentence, then launch -ready work. Explain interpreter, routing-gate, or adapter details only +Before delegating, describe the task split in at most one short sentence, then +launch ready work. Explain adapter details only when requested or needed to explain a failure or blocker. Keep later updates focused on findings, blockers, and results. diff --git a/skills/smart-router/SKILL.md b/skills/smart-router/SKILL.md index defa78046..f1a5b7a7d 100644 --- a/skills/smart-router/SKILL.md +++ b/skills/smart-router/SKILL.md @@ -27,7 +27,7 @@ or first prompt. In sessions launched with `ENABLE_ORCHESTRATION=1`, it also con orchestration. When turned off, earlier orchestrate instructions are superseded: do not start new automatic delegation or fall back to orchestrator role models. Continue in the root unless the user explicitly requests a subagent. Honor that request using -native tools and normal harness model selection, without the orchestrator or its resolution -helper; keep routing off. Existing children can finish. When turned on, apply the orchestrate +native tools and normal harness model selection, without the orchestrator; keep routing off. +Existing children can finish. When turned on, apply the orchestrate skill to further work only if the session was launched with `ENABLE_ORCHESTRATION=1`. Do not change the orchestration feature flag. Return the command's result. diff --git a/src/ucode/smart_routing/orchestrator.py b/src/ucode/smart_routing/orchestrator.py index e79a8410d..98890e23b 100644 --- a/src/ucode/smart_routing/orchestrator.py +++ b/src/ucode/smart_routing/orchestrator.py @@ -19,11 +19,9 @@ HOOK_MODULE = "ucode.smart_routing.orchestrator" DISABLED_CONTEXT = ( - "UG automatic orchestration is off for this session. " - "This supersedes any earlier model-orchestrator workflow: do not start new automatic " - "delegation or fall back to orchestrator role models. Explicit user requests for subagents " - "still use native tools and the current Smart Router setting, without the orchestrator " - "or its routing check. Otherwise continue the task in the root. " + "UG automatic orchestration is off for this session. This supersedes earlier " + "orchestrator instructions. Continue in the root unless the user explicitly requests " + "subagents; use native tools and the current Smart Router setting for those requests. " "Collect results from children already running." ) @@ -50,11 +48,6 @@ def enabled(env: Mapping[str, str] | None = None) -> bool: return smart_routing_enabled(effective_environment(source)) -def require_enabled() -> None: - if not enabled(): - raise ValueError(DISABLED_CONTEXT) - - def skill_directory() -> Path: return skills._skills_source() / skills.ORCHESTRATOR_SKILL @@ -99,27 +92,14 @@ def hook_output(payload: object) -> dict | None: except (OSError, UnicodeError): return None context = ( - "Apply the UG model-orchestrator workflow to this task. " - "Orchestration is opted in and follows Smart Router's session controls.\n" + "UG automatic orchestration is on for this session. Apply the workflow below.\n" f"Skill directory: {directory}\n\n{workflow}" ) return {"hookSpecificOutput": {"hookEventName": event, "additionalContext": context}} -def main(argv: list[str] | None = None) -> None: - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument( - "--check", - action="store_true", - help="Exit successfully only in an opted-in, enabled smart-routing session.", - ) - if parser.parse_args(argv).check: - try: - require_enabled() - except ValueError as exc: - parser.exit(1, f"{exc}\n") - return - +def main() -> None: + argparse.ArgumentParser(description=__doc__).parse_args() try: payload = json.load(sys.stdin) except (OSError, UnicodeError, ValueError): diff --git a/src/ucode/smart_routing/v2.py b/src/ucode/smart_routing/v2.py index 467f1fa6c..964583118 100644 --- a/src/ucode/smart_routing/v2.py +++ b/src/ucode/smart_routing/v2.py @@ -663,7 +663,7 @@ def launch_codex( overlay["features.hooks"] = True _prepare_smart_router_session("codex") # Codex constructs tool subprocess environments through its shell policy. - # The skill's gate needs the same launch baseline as the routing hook, even + # The hooks and Smart Router toggle need the same launch baseline, even # when the user's policy filters inherited environment variables. for key in (SESSION_ENV_VAR, SESSION_PYTHON_ENV_VAR, *SMART_ROUTING_ENV_KEYS): if key in os.environ: diff --git a/tests/README.md b/tests/README.md index ee3fe8463..6dd429d50 100644 --- a/tests/README.md +++ b/tests/README.md @@ -135,11 +135,7 @@ The toggle integration journeys run with `ENABLE_ORCHESTRATION` unset and with `ENABLE_ORCHESTRATION=1`. They require only `smart-router` by default and both bundled skills when opted in, verify the saved session controls and native tool-result confirmation after each toggle, and explicitly request their children, -including while routing is off. Claude's journey -answers the visible permission prompt for the exact read-only orchestrator check, -using the dialog's command because pending calls may not yet be in the transcript. -Evidence tests reject added shell commands, commands in scrollback or descriptions, -and other permission selections. +including while routing is off. `test_integration_evidence.py` checks native tool-result extraction for both agents, including collapsed-output records, and excludes user echoes and assistant claims. diff --git a/tests/integration/README.md b/tests/integration/README.md index 92b40cdde..cd9c29cc7 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -295,12 +295,6 @@ eligibility, role-contract preservation, and isolation from legacy preference fi lack dedicated regression coverage. Codex's native hook merging, project trust, and execution of pre-existing hooks are not exercised by this integration suite. -The Claude on/off/on journey handles the visible permission prompt for the exact -read-only orchestrator check. It verifies the command in the current dialog and -waits for the dialog to settle before selecting the one-time Yes option. Pending -calls need not appear in the native transcript until approval. It still requires the routed -banner, completed child, and correlated routing decision. - The portable `../test_claude_windows_smart_routing.py` checks the Windows subagent-only fallback without Unix imports. Native Windows TUI and hook execution remain outside this integration suite. diff --git a/tests/integration/test_ug_smart_routing_hooks.py b/tests/integration/test_ug_smart_routing_hooks.py index 187fb26b0..1cd6ffdf5 100644 --- a/tests/integration/test_ug_smart_routing_hooks.py +++ b/tests/integration/test_ug_smart_routing_hooks.py @@ -19,7 +19,6 @@ assert_subagent_routed, assistant_answers, is_child_session, - is_orchestrator_check_permission, read_jsonl, tool_outputs, ) @@ -91,29 +90,8 @@ def _run_calculation(tui, session, agent: str, expression: str, expected: str, * tui.submit(task.prompt) if routed: - permission_in_progress = False - - def routed_banner_visible(screen): - nonlocal permission_in_progress - if "Do you want to proceed?" in screen: - if permission_in_progress: - return False - if agent != "claude" or not is_orchestrator_check_permission(screen): - return False - # Claude briefly ignores input when a permission dialog opens. - tui.wait_for( - is_orchestrator_check_permission, - "the read-only orchestrator permission dialog to settle", - timeout=5, - ) - tui.send("\r", "allow the read-only orchestrator routing-state check") - permission_in_progress = True - return False - permission_in_progress = False - return _routing_banner_for_task(screen, task.marker) - tui.wait_for( - routed_banner_visible, + lambda screen: _routing_banner_for_task(screen, task.marker), f"the Smart Router subagent banner for {task.marker}", timeout=120, ) @@ -322,8 +300,7 @@ def test_smart_router_skill_toggles_claude_subagent_routing( routing controls, even with collapsed terminal output; all three uniquely tagged calculations complete in native child sessions; only the first and third show the subagent-routing banner and produce live gateway decisions correlated with those children. - A visible permission prompt for the exact read-only orchestrator check is accepted; - other commands are rejected. No first-prompt routing wrapper starts. + No first-prompt routing wrapper starts. """ session = live_session session.env["TMPDIR"] = str(tmp_path) diff --git a/tests/integration/utils/evidence.py b/tests/integration/utils/evidence.py index fbaa7f5d2..d3f77a893 100644 --- a/tests/integration/utils/evidence.py +++ b/tests/integration/utils/evidence.py @@ -88,26 +88,6 @@ def is_child_session(agent: str, path: str, records: list[dict]) -> bool: return _AGENT_HELPERS.get(agent, codex).is_child_session(path, records) -def is_orchestrator_check_permission(screen: str) -> bool: - """Recognize one-time approval for the exact read-only orchestrator check.""" - command = '"$UCODE_SMART_ROUTER_PYTHON" -m ucode.smart_routing.orchestrator --check' - # Pending calls need not reach the transcript until approval. Inspect the - # current dialog's first command line; multiline commands have a │ gutter. - dialog = re.split(r"(?m)^[ \t]*─{3,}[ \t]*$", screen)[-1] - return bool( - re.match( - r"\s*Bash command[ \t]*\n" - r"(?:[ \t]*Tip:[^\n]*\n)?" - r"[ \t]*\n" - rf"[ \t]*{re.escape(command)}[ \t]*\n", - dialog, - ) - and re.search(r"(?m)^[ \t]*Do you want to proceed\?[ \t]*$", dialog) - and re.search(r"(?m)^[ \t]*[›❯>][ \t]*1\.[ \t]*Yes[ \t]*$", dialog) - and re.search(r"(?m)^[ \t]*Esc to cancel\b", dialog) - ) - - def completed_task_models(session, agent: str, answer_value: str) -> set[str]: """Read parent task model evidence, not proof of the gateway's destination.""" assert agent in _AGENT_HELPERS, f"Unsupported evidence agent: {agent}" diff --git a/tests/test_integration_evidence.py b/tests/test_integration_evidence.py index a440af7b2..cf04efc76 100644 --- a/tests/test_integration_evidence.py +++ b/tests/test_integration_evidence.py @@ -85,49 +85,6 @@ def test_transient_retries_and_running_tasks_are_not_terminal_errors(screen): assert_no_terminal_api_error(screen) -@pytest.mark.parametrize( - "command_suffix,title,selected,accepted", - [ - ("", "Bash command", "1. Yes", True), - ("; echo unrelated", "Bash command", "1. Yes", False), - ("\necho unrelated", "Bash command", "1. Yes", False), - ("", "Read file", "1. Yes", False), - ("", "Bash command", "2. Yes, and switch to auto mode", False), - ("", "Bash command", "3. No", False), - ], -) -@pytest.mark.parametrize("tip", ["", " Tip: auto mode handles these prompts for you\n"]) -def test_orchestrator_permission_requires_exact_visible_command( - command_suffix, title, selected, accepted, tip -): - command = '"$UCODE_SMART_ROUTER_PYTHON" -m ucode.smart_routing.orchestrator --check' - displayed = command + command_suffix - if "\n" in displayed: - displayed = "\n".join("│ " + line for line in displayed.splitlines()) - screen = ( - "Earlier tool output\n" + "─" * 80 + "\n" - f" {title}\n{tip}\n" - f" {displayed}\n Check smart routing gate status\n\n" - f" Contains simple_expansion\n\n Do you want to proceed?\n ❯ {selected}\n\n" - " Esc to cancel · Tab to amend\n" - ) - - assert evidence.is_orchestrator_check_permission(screen) is accepted - - -@pytest.mark.parametrize("command_location", ["scrollback", "description"]) -def test_orchestrator_command_outside_dialog_command_does_not_grant_permission(command_location): - command = '"$UCODE_SMART_ROUTER_PYTHON" -m ucode.smart_routing.orchestrator --check' - screen = ( - f"{command if command_location == 'scrollback' else ''}\n" + "─" * 80 + "\n" - " Bash command\n\n echo unrelated\n" - f" {command if command_location == 'description' else 'An unrelated command'}\n\n" - " Do you want to proceed?\n ❯ 1. Yes\n\n Esc to cancel · Tab to amend\n" - ) - - assert not evidence.is_orchestrator_check_permission(screen) - - @pytest.mark.parametrize("agent", ["claude", "codex"]) def test_tagged_calculation_requires_the_native_child_answer(tmp_path, agent): session = _Session(tmp_path) From dafbac25eb155e0d9065e7fc2ebc55b49e02ee78 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 19:28:44 +0000 Subject: [PATCH 14/25] Name the bundled workflow Smart Router Orchestrator --- README.md | 9 +++++---- skills/orchestrate/README.md | 6 +++--- skills/orchestrate/SKILL.md | 4 ++-- skills/smart-router/SKILL.md | 6 +++--- src/ucode/smart_routing/orchestrator.py | 6 +++--- 5 files changed, 16 insertions(+), 15 deletions(-) diff --git a/README.md b/README.md index acbf4441b..9fc527ec9 100644 --- a/README.md +++ b/README.md @@ -246,11 +246,12 @@ is enabled, ug warns and falls back to subagent routing because the first-prompt wrapper requires a Unix terminal. The generated shell hooks expect Git Bash; PowerShell-only setups are not covered. -### Automatic orchestration +### Smart Router Orchestrator Smart-routed Claude and Codex sessions install `smart-router`. Set -`ENABLE_ORCHESTRATION=1` at launch to also install and activate the bundled -`orchestrate` skill; orchestration is off by default. For example: +`ENABLE_ORCHESTRATION=1` at launch to also install and activate Smart Router +Orchestrator through the bundled `orchestrate` skill; orchestration is off by +default. For example: ```bash ENABLE_ORCHESTRATION=1 ENABLE_SMART_ROUTING_SUBAGENT_ONLY=1 ug claude @@ -276,7 +277,7 @@ routing still checks the controls for each subagent. UG supplies its own hooks; Codex combines them with existing hooks and applies project trust. Smart routing selects subagent models; separate role-model preferences are ignored and their files are left untouched. See the bundled -[orchestrator documentation](skills/orchestrate/README.md) for details. +[Smart Router Orchestrator documentation](skills/orchestrate/README.md) for details. ## Managed Files diff --git a/skills/orchestrate/README.md b/skills/orchestrate/README.md index d39950757..7a31e0823 100644 --- a/skills/orchestrate/README.md +++ b/skills/orchestrate/README.md @@ -1,7 +1,7 @@ -# UG model orchestrator +# Smart Router Orchestrator -UG bundles the `orchestrate` workflow and five Claude role definitions from -`model-orchestrator` 0.4.10. Smart-routed Claude and Codex launches install and +UG bundles the `orchestrate` workflow and five Claude role definitions. +Smart-routed Claude and Codex launches install and activate this skill alongside `smart-router` only with `ENABLE_ORCHESTRATION=1`. The feature is off by default; routing alone installs only `smart-router`. diff --git a/skills/orchestrate/SKILL.md b/skills/orchestrate/SKILL.md index 99b2441bf..7143fc9d4 100644 --- a/skills/orchestrate/SKILL.md +++ b/skills/orchestrate/SKILL.md @@ -1,13 +1,13 @@ --- name: orchestrate -description: Coordinate substantive development with native subagents when ENABLE_ORCHESTRATION=1 and Unity Gateway smart routing is enabled. Follow UG's activation context. Skip easy tasks and explicit no-subagent requests. +description: Smart Router Orchestrator coordinates substantive development with native subagents when ENABLE_ORCHESTRATION=1 and Unity Gateway smart routing is enabled. Follow UG's activation context. Skip easy tasks and explicit no-subagent requests. model: inherit argument-hint: "[task]" metadata: version: "1.1.0" --- -# Model orchestrator +# Smart Router Orchestrator ## Activation diff --git a/skills/smart-router/SKILL.md b/skills/smart-router/SKILL.md index f1a5b7a7d..ace674eed 100644 --- a/skills/smart-router/SKILL.md +++ b/skills/smart-router/SKILL.md @@ -1,6 +1,6 @@ --- name: smart-router -description: Enable or disable Unity Gateway subagent model routing for the current smart-routed Claude or Codex session, along with automatic orchestration when opted in. +description: Enable or disable Unity Gateway subagent model routing for the current smart-routed Claude or Codex session, along with Smart Router Orchestrator when opted in. allowed-tools: Bash("$UCODE_SMART_ROUTER_PYTHON" -m ucode.cli claude --enable-smart-routing), Bash("$UCODE_SMART_ROUTER_PYTHON" -m ucode.cli claude --disable-smart-routing), Bash("$UCODE_SMART_ROUTER_PYTHON" -m ucode.cli codex --enable-smart-routing), Bash("$UCODE_SMART_ROUTER_PYTHON" -m ucode.cli codex --disable-smart-routing) metadata: version: "1.2.0" @@ -23,8 +23,8 @@ to restart through an updated Unity Gateway with smart routing enabled. With no argument, explain that only `on` and `off` are accepted. Do not edit the state file. This affects subsequent subagent model selection in the current session, not the root model -or first prompt. In sessions launched with `ENABLE_ORCHESTRATION=1`, it also controls automatic -orchestration. When turned off, earlier orchestrate instructions +or first prompt. In sessions launched with `ENABLE_ORCHESTRATION=1`, it also controls Smart Router +Orchestrator. When turned off, earlier orchestrate instructions are superseded: do not start new automatic delegation or fall back to orchestrator role models. Continue in the root unless the user explicitly requests a subagent. Honor that request using native tools and normal harness model selection, without the orchestrator; keep routing off. diff --git a/src/ucode/smart_routing/orchestrator.py b/src/ucode/smart_routing/orchestrator.py index 98890e23b..937774f67 100644 --- a/src/ucode/smart_routing/orchestrator.py +++ b/src/ucode/smart_routing/orchestrator.py @@ -1,4 +1,4 @@ -"""Activate the bundled orchestrator in opted-in smart-routing sessions.""" +"""Activate Smart Router Orchestrator in opted-in smart-routing sessions.""" from __future__ import annotations @@ -19,7 +19,7 @@ HOOK_MODULE = "ucode.smart_routing.orchestrator" DISABLED_CONTEXT = ( - "UG automatic orchestration is off for this session. This supersedes earlier " + "Smart Router Orchestrator is off for this session. This supersedes earlier " "orchestrator instructions. Continue in the root unless the user explicitly requests " "subagents; use native tools and the current Smart Router setting for those requests. " "Collect results from children already running." @@ -92,7 +92,7 @@ def hook_output(payload: object) -> dict | None: except (OSError, UnicodeError): return None context = ( - "UG automatic orchestration is on for this session. Apply the workflow below.\n" + "Smart Router Orchestrator is on for this session. Apply the workflow below.\n" f"Skill directory: {directory}\n\n{workflow}" ) return {"hookSpecificOutput": {"hookEventName": event, "additionalContext": context}} From 5ab64e88ab5865784883911168fb6a850393bdb0 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 20:33:30 +0000 Subject: [PATCH 15/25] Align CUJ2 fixture with published managed skill --- tests/README.md | 4 ++-- tests/e2e_cuj/helpers/constants.py | 1 + tests/e2e_cuj/test_cuj2_mps_mcp.py | 9 ++++++--- tests/integration/README.md | 4 +++- 4 files changed, 12 insertions(+), 6 deletions(-) diff --git a/tests/README.md b/tests/README.md index 6dd429d50..bab9a2923 100644 --- a/tests/README.md +++ b/tests/README.md @@ -169,7 +169,7 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | Test | User action | Expected evidence | | --- | --- | --- | -| `test_cuj_configuration`, `test_cuj_codex_inference`, `test_cuj_claude_inference` in `test_cuj2_mps_mcp.py` | Read the permanently preconfigured two-agent MPS/MCP config; run separate configure, Codex TUI, and Claude TUI cases | Configuration verifies both native MPS APIs, generated settings, and sandbox inclusion with web_search excluded; each TUI case records its exact MPS target header and model, paired HTTP 200 response, and final marker. Live sandbox execution is deferred because MAS cannot downscope the CI service principal. No workspace config CRUD is performed | +| `test_cuj_configuration`, `test_cuj_codex_inference`, `test_cuj_claude_inference` in `test_cuj2_mps_mcp.py` | Read the permanently preconfigured two-agent MPS/MCP config and named-skill selector; run separate configure, Codex TUI, and Claude TUI cases | Configuration verifies both native MPS APIs, generated settings, and sandbox inclusion with web_search excluded; each TUI case records its exact MPS target header and model, paired HTTP 200 response, and final marker. Live sandbox execution is deferred because MAS cannot downscope the CI service principal. No workspace config CRUD is performed | | `test_ug_configure_claude_databricks` | Configure Databricks Hosted; execute the generated auth helper; launch plain `ug claude`, read a file, and open `/model` | Generated helper invokes `ug` with clean token stdout; assistant returns an unpredictable file value; native discovery caches `system.ai` models and the picker shows a discovered model without an opt-in flag; normal exit; reopen with working keyboard input | | `test_ug_configure_claude_anthropic_mps` | Select Anthropic MPS in the real configure picker; launch Claude | Saved provider in status; completed TUI file task; normal exit | | `test_ug_configure_codex_databricks` | Configure Databricks Hosted; execute the generated auth helper; open Codex TUI and read a file | Generated helper invokes `ug` with clean token stdout; completed assistant answer contains the file value; normal exit and reopen | @@ -342,7 +342,7 @@ These are unit/component checks; they do not establish live sudo password-prompt | Scenario | Status / requirement | | --- | --- | | Broad live MCP functionality | CUJ2 covers sandbox MCP configuration and generated client listings. Live sandbox execution is deferred because MAS cannot downscope the CI service principal. Installation tests cover the local web-search MCP handshake and tool listing, not upstream proxying or a real search request. Other MCP services, live parent/child search, and permission decisions remain deferred | -| Dedicated CUJ Skill coverage | Deferred because of a backend storage-path issue; CUJ2 has no Skill configuration, download, invocation, or assertion path | +| Dedicated CUJ Skill coverage | CUJ2 requires `skills.names = ["ug_e2e.skills.fixture-summary"]` in the published configuration. Skill download, discovery, and invocation are not asserted by these cases | | Broad configure flags, multiple workspaces, and PAT flows | Deferred while focusing on basic CUJs | | Workspace-switch MCP cleanup | The `workspace_switch` CUJ covers real registration, cleanup, repeat configure, and a completed Claude task. Unit/component tests cover duplicate attempts and injected removal failures; the CUJ does not force an agent timeout. It runs in the required managed CI lane for full/live runs. | | Relayed/subscription MPS discovery | Not covered by the scoped discovery journeys | diff --git a/tests/e2e_cuj/helpers/constants.py b/tests/e2e_cuj/helpers/constants.py index e05da54ff..60a24a2b4 100644 --- a/tests/e2e_cuj/helpers/constants.py +++ b/tests/e2e_cuj/helpers/constants.py @@ -41,3 +41,4 @@ class CodingAgent(StrEnum): ) UC_MODEL_LOCATION_FIXTURE = ("ug_e2e.models", "ug_e2e.models.codex_primary") FIXTURE_READER_MCP_SERVICE_NAME = "ug_e2e.tools.fixture_reader" +FIXTURE_SUMMARY_SKILL_NAME = "ug_e2e.skills.fixture-summary" diff --git a/tests/e2e_cuj/test_cuj2_mps_mcp.py b/tests/e2e_cuj/test_cuj2_mps_mcp.py index 0af6311df..c18d90be0 100644 --- a/tests/e2e_cuj/test_cuj2_mps_mcp.py +++ b/tests/e2e_cuj/test_cuj2_mps_mcp.py @@ -22,6 +22,7 @@ CLAUDE, CODEX, CODING_AGENT_BY_CLI_NAME, + FIXTURE_SUMMARY_SKILL_NAME, INFERENCE_PATHS, MODEL_PROVIDER_SERVICE_FIXTURES, SANDBOX_MCP_SERVICE_NAME, @@ -53,6 +54,7 @@ def _assert_cuj2_config(config: dict) -> None: "default_agent": CODING_AGENT_BY_CLI_NAME[CODEX], "enabled_agents": [expected_agent_config(CLAUDE), expected_agent_config(CODEX)], "mcp_servers": {"names": [SANDBOX_MCP_SERVICE_NAME]}, + "skills": {"names": [FIXTURE_SUMMARY_SKILL_NAME]}, } assert actual == expected, config @@ -203,9 +205,10 @@ class TestCuj2Configuration(_Cuj2Base): def test_cuj_configuration(self, cuj): """Scenario: configure ug from the preconfigured two-agent MPS/MCP workspace. - Expected: the read-only CodingAgentConfig selects the two exact MPS resources and sandbox - MCP; both native MPS APIs advertise their selected model, generated agent settings use - the exact provider/model values, and agent MCP listings exclude web_search. + Expected: the read-only CodingAgentConfig selects the two exact MPS resources, sandbox + MCP, and fixture-summary skill; both native MPS APIs advertise their selected model, + generated agent settings use the exact provider/model values, and agent MCP listings + exclude web_search. Skill download and invocation are not asserted here. """ session, workspace, _ = cuj published = workspace.config() diff --git a/tests/integration/README.md b/tests/integration/README.md index cd9c29cc7..01566629f 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -7,7 +7,9 @@ Its Claude/Codex evidence helpers keep scenario-specific assertions separate fro Workspace config/catalog reads use its base class's Databricks SDK client. Configuration is read-only and checked for changes at teardown; concurrent readers need no reservation. CUJ2 adds three separately collected cases for exact MPS/MCP configuration, Codex inference, -and Claude inference. +and Claude inference. Its published configuration must include +`skills.names = ["ug_e2e.skills.fixture-summary"]`; these cases do not assert skill download, +discovery, or invocation. The [catalog discovery journey](../e2e_cuj/README.md) uses the CUJ3 workspace to check agent-compatible pickers, schema exclusions, configured defaults, and real inference. From 94feca72c56aa50a8d29c5d08ecaa4524be467e8 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 20:38:57 +0000 Subject: [PATCH 16/25] Remove unrelated CUJ fixture changes --- tests/README.md | 4 ++-- tests/e2e_cuj/helpers/constants.py | 1 - tests/e2e_cuj/test_cuj2_mps_mcp.py | 9 +++------ tests/integration/README.md | 4 +--- 4 files changed, 6 insertions(+), 12 deletions(-) diff --git a/tests/README.md b/tests/README.md index bab9a2923..6dd429d50 100644 --- a/tests/README.md +++ b/tests/README.md @@ -169,7 +169,7 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | Test | User action | Expected evidence | | --- | --- | --- | -| `test_cuj_configuration`, `test_cuj_codex_inference`, `test_cuj_claude_inference` in `test_cuj2_mps_mcp.py` | Read the permanently preconfigured two-agent MPS/MCP config and named-skill selector; run separate configure, Codex TUI, and Claude TUI cases | Configuration verifies both native MPS APIs, generated settings, and sandbox inclusion with web_search excluded; each TUI case records its exact MPS target header and model, paired HTTP 200 response, and final marker. Live sandbox execution is deferred because MAS cannot downscope the CI service principal. No workspace config CRUD is performed | +| `test_cuj_configuration`, `test_cuj_codex_inference`, `test_cuj_claude_inference` in `test_cuj2_mps_mcp.py` | Read the permanently preconfigured two-agent MPS/MCP config; run separate configure, Codex TUI, and Claude TUI cases | Configuration verifies both native MPS APIs, generated settings, and sandbox inclusion with web_search excluded; each TUI case records its exact MPS target header and model, paired HTTP 200 response, and final marker. Live sandbox execution is deferred because MAS cannot downscope the CI service principal. No workspace config CRUD is performed | | `test_ug_configure_claude_databricks` | Configure Databricks Hosted; execute the generated auth helper; launch plain `ug claude`, read a file, and open `/model` | Generated helper invokes `ug` with clean token stdout; assistant returns an unpredictable file value; native discovery caches `system.ai` models and the picker shows a discovered model without an opt-in flag; normal exit; reopen with working keyboard input | | `test_ug_configure_claude_anthropic_mps` | Select Anthropic MPS in the real configure picker; launch Claude | Saved provider in status; completed TUI file task; normal exit | | `test_ug_configure_codex_databricks` | Configure Databricks Hosted; execute the generated auth helper; open Codex TUI and read a file | Generated helper invokes `ug` with clean token stdout; completed assistant answer contains the file value; normal exit and reopen | @@ -342,7 +342,7 @@ These are unit/component checks; they do not establish live sudo password-prompt | Scenario | Status / requirement | | --- | --- | | Broad live MCP functionality | CUJ2 covers sandbox MCP configuration and generated client listings. Live sandbox execution is deferred because MAS cannot downscope the CI service principal. Installation tests cover the local web-search MCP handshake and tool listing, not upstream proxying or a real search request. Other MCP services, live parent/child search, and permission decisions remain deferred | -| Dedicated CUJ Skill coverage | CUJ2 requires `skills.names = ["ug_e2e.skills.fixture-summary"]` in the published configuration. Skill download, discovery, and invocation are not asserted by these cases | +| Dedicated CUJ Skill coverage | Deferred because of a backend storage-path issue; CUJ2 has no Skill configuration, download, invocation, or assertion path | | Broad configure flags, multiple workspaces, and PAT flows | Deferred while focusing on basic CUJs | | Workspace-switch MCP cleanup | The `workspace_switch` CUJ covers real registration, cleanup, repeat configure, and a completed Claude task. Unit/component tests cover duplicate attempts and injected removal failures; the CUJ does not force an agent timeout. It runs in the required managed CI lane for full/live runs. | | Relayed/subscription MPS discovery | Not covered by the scoped discovery journeys | diff --git a/tests/e2e_cuj/helpers/constants.py b/tests/e2e_cuj/helpers/constants.py index 60a24a2b4..e05da54ff 100644 --- a/tests/e2e_cuj/helpers/constants.py +++ b/tests/e2e_cuj/helpers/constants.py @@ -41,4 +41,3 @@ class CodingAgent(StrEnum): ) UC_MODEL_LOCATION_FIXTURE = ("ug_e2e.models", "ug_e2e.models.codex_primary") FIXTURE_READER_MCP_SERVICE_NAME = "ug_e2e.tools.fixture_reader" -FIXTURE_SUMMARY_SKILL_NAME = "ug_e2e.skills.fixture-summary" diff --git a/tests/e2e_cuj/test_cuj2_mps_mcp.py b/tests/e2e_cuj/test_cuj2_mps_mcp.py index c18d90be0..0af6311df 100644 --- a/tests/e2e_cuj/test_cuj2_mps_mcp.py +++ b/tests/e2e_cuj/test_cuj2_mps_mcp.py @@ -22,7 +22,6 @@ CLAUDE, CODEX, CODING_AGENT_BY_CLI_NAME, - FIXTURE_SUMMARY_SKILL_NAME, INFERENCE_PATHS, MODEL_PROVIDER_SERVICE_FIXTURES, SANDBOX_MCP_SERVICE_NAME, @@ -54,7 +53,6 @@ def _assert_cuj2_config(config: dict) -> None: "default_agent": CODING_AGENT_BY_CLI_NAME[CODEX], "enabled_agents": [expected_agent_config(CLAUDE), expected_agent_config(CODEX)], "mcp_servers": {"names": [SANDBOX_MCP_SERVICE_NAME]}, - "skills": {"names": [FIXTURE_SUMMARY_SKILL_NAME]}, } assert actual == expected, config @@ -205,10 +203,9 @@ class TestCuj2Configuration(_Cuj2Base): def test_cuj_configuration(self, cuj): """Scenario: configure ug from the preconfigured two-agent MPS/MCP workspace. - Expected: the read-only CodingAgentConfig selects the two exact MPS resources, sandbox - MCP, and fixture-summary skill; both native MPS APIs advertise their selected model, - generated agent settings use the exact provider/model values, and agent MCP listings - exclude web_search. Skill download and invocation are not asserted here. + Expected: the read-only CodingAgentConfig selects the two exact MPS resources and sandbox + MCP; both native MPS APIs advertise their selected model, generated agent settings use + the exact provider/model values, and agent MCP listings exclude web_search. """ session, workspace, _ = cuj published = workspace.config() diff --git a/tests/integration/README.md b/tests/integration/README.md index 01566629f..cd9c29cc7 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -7,9 +7,7 @@ Its Claude/Codex evidence helpers keep scenario-specific assertions separate fro Workspace config/catalog reads use its base class's Databricks SDK client. Configuration is read-only and checked for changes at teardown; concurrent readers need no reservation. CUJ2 adds three separately collected cases for exact MPS/MCP configuration, Codex inference, -and Claude inference. Its published configuration must include -`skills.names = ["ug_e2e.skills.fixture-summary"]`; these cases do not assert skill download, -discovery, or invocation. +and Claude inference. The [catalog discovery journey](../e2e_cuj/README.md) uses the CUJ3 workspace to check agent-compatible pickers, schema exclusions, configured defaults, and real inference. From 9284679c2b7bca494f9998a133ab772b898a8c30 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 20:47:46 +0000 Subject: [PATCH 17/25] Drop nested-launch session reset from orchestrator integration --- src/ucode/cli.py | 8 ++------ src/ucode/smart_routing/session_env.py | 18 +----------------- tests/README.md | 2 +- tests/integration/README.md | 4 ++-- 4 files changed, 6 insertions(+), 26 deletions(-) diff --git a/src/ucode/cli.py b/src/ucode/cli.py index d2acd26bb..656abd746 100644 --- a/src/ucode/cli.py +++ b/src/ucode/cli.py @@ -139,7 +139,6 @@ from ucode.smart_routing.claude_hooks import FIRST_PROMPT_SOCKET_ENV, ROUTE_FIRST_PROMPT_EVENT from ucode.smart_routing.session_env import ( effective_environment, - fresh_launch, session_env_path, set_session_environment, ) @@ -2938,11 +2937,8 @@ def _launch_tool( provider=provider, ) print_success(f"Starting {TOOL_SPECS[tool]['display']}") - with ( - _smart_routing_v2_flag( - True if managed_smart_routing_enabled and smart_routing_enabled else None - ), - fresh_launch(), + with _smart_routing_v2_flag( + True if managed_smart_routing_enabled and smart_routing_enabled else None ): launch_agent(tool, state, ctx.args, options=launch_options) except RuntimeError as exc: diff --git a/src/ucode/smart_routing/session_env.py b/src/ucode/smart_routing/session_env.py index a134e460a..21887c945 100644 --- a/src/ucode/smart_routing/session_env.py +++ b/src/ucode/smart_routing/session_env.py @@ -6,8 +6,7 @@ import os import sys import tempfile -from collections.abc import Iterator, Mapping, MutableMapping -from contextlib import contextmanager +from collections.abc import Mapping, MutableMapping from pathlib import Path from ucode.config_io import atomic_write_json @@ -18,21 +17,6 @@ _ALLOWED_KEYS = frozenset(SMART_ROUTING_ENV_KEYS) -@contextmanager -def fresh_launch() -> Iterator[None]: - """Prevent a nested, non-routed launch from inheriting its parent's eligibility.""" - keys = (SESSION_ENV_VAR, SESSION_PYTHON_ENV_VAR) - previous = {key: os.environ.pop(key, None) for key in keys} - try: - yield - finally: - for key, value in previous.items(): - if value is None: - os.environ.pop(key, None) - else: - os.environ[key] = value - - def start_session(env: MutableMapping[str, str] | None = None) -> Path: """Create an empty override file and expose it to the launched harness.""" target = os.environ if env is None else env diff --git a/tests/README.md b/tests/README.md index 6dd429d50..d78d3e368 100644 --- a/tests/README.md +++ b/tests/README.md @@ -140,7 +140,7 @@ including while routing is off. including collapsed-output records, and excludes user echoes and assistant claims. Dedicated regression coverage is missing for root-only orchestrator activation, -compaction, retained skills in ineligible sessions, nested-session eligibility, +compaction, retained skills in ineligible sessions, role-contract preservation, and isolation from legacy preference files. Codex's native hook merging and project trust, automatic delegation, and execution of pre-existing hooks remain outside the integration suite. diff --git a/tests/integration/README.md b/tests/integration/README.md index cd9c29cc7..73a2736cd 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -290,8 +290,8 @@ Collapsed terminal output is allowed; the answer need not repeat the CLI's exact Each following child still verifies whether a routing decision occurred. Their off-phase child is an explicit user-requested delegation; these journeys do not establish automatic orchestration behavior. Root-only -activation, compaction, retained skills in ineligible sessions, nested-session -eligibility, role-contract preservation, and isolation from legacy preference files +activation, compaction, retained skills in ineligible sessions, +role-contract preservation, and isolation from legacy preference files lack dedicated regression coverage. Codex's native hook merging, project trust, and execution of pre-existing hooks are not exercised by this integration suite. From d834a4effadc0ce85e9cc0d01ef66017c7537605 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 20:57:30 +0000 Subject: [PATCH 18/25] Use Codex's default environment inheritance --- src/ucode/smart_routing/v2.py | 15 ++++++--------- 1 file changed, 6 insertions(+), 9 deletions(-) diff --git a/src/ucode/smart_routing/v2.py b/src/ucode/smart_routing/v2.py index 964583118..8db8a1bff 100644 --- a/src/ucode/smart_routing/v2.py +++ b/src/ucode/smart_routing/v2.py @@ -661,16 +661,13 @@ def launch_codex( overlay["model_catalog_json"] = str(catalog_path) overlay["hooks"] = _v2_hooks(state, available_models) overlay["features.hooks"] = True - _prepare_smart_router_session("codex") + session_env_path = _prepare_smart_router_session("codex") # Codex constructs tool subprocess environments through its shell policy. - # The hooks and Smart Router toggle need the same launch baseline, even - # when the user's policy filters inherited environment variables. - for key in (SESSION_ENV_VAR, SESSION_PYTHON_ENV_VAR, *SMART_ROUTING_ENV_KEYS): - if key in os.environ: - overlay[f"shell_environment_policy.set.{key}"] = os.environ[key] - overlay[f"shell_environment_policy.set.{ENABLE_ORCHESTRATION_ENV_VAR}"] = ( - "1" if orchestrator.feature_enabled() else "0" - ) + # Pass both the session marker and its launching interpreter through that policy. + overlay[f"shell_environment_policy.set.{SESSION_ENV_VAR}"] = str(session_env_path) + overlay[f"shell_environment_policy.set.{SESSION_PYTHON_ENV_VAR}"] = os.environ[ + SESSION_PYTHON_ENV_VAR + ] config_args = codex_config_args(overlay) if not first_prompt_routing_enabled(): # Subagent-only routing needs neither the app-server nor the interposer: From b0af05a723e9be1942d8d6fc114bf168e8093ada Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 22:08:59 +0000 Subject: [PATCH 19/25] Test automatic orchestration and post-compaction recovery --- tests/README.md | 15 +- tests/integration/README.md | 21 ++- tests/integration/test_ug_orchestrator.py | 206 ++++++++++++++++++++++ tests/integration/utils/evidence.py | 52 ++++++ tests/test_integration_evidence.py | 85 +++++++++ 5 files changed, 369 insertions(+), 10 deletions(-) create mode 100644 tests/integration/test_ug_orchestrator.py diff --git a/tests/README.md b/tests/README.md index d78d3e368..7d1a687d7 100644 --- a/tests/README.md +++ b/tests/README.md @@ -139,11 +139,15 @@ including while routing is off. `test_integration_evidence.py` checks native tool-result extraction for both agents, including collapsed-output records, and excludes user echoes and assistant claims. -Dedicated regression coverage is missing for root-only orchestrator activation, -compaction, retained skills in ineligible sessions, -role-contract preservation, and isolation from legacy preference files. Codex's -native hook merging and project trust, automatic delegation, and execution of -pre-existing hooks remain outside the integration suite. +`test_ug_orchestrator.py` requests a review of three source modules without mentioning +subagents. It requires a completed native child response and a root report containing +traceability values read from all three files. Each harness then performs native +`/compact`, receives the full workflow again, and completes a second review with new +child work. The journeys check orchestration and continuation, not review accuracy. +Retained skills in ineligible sessions, role-contract preservation, and isolation +from legacy preference files still lack dedicated regression coverage. Codex's +native hook merging and project trust, and execution of pre-existing hooks remain +outside the integration suite. The portable Windows routing test checks native executable forwarding, generated hooks/plugins, caller arguments, and cleanup without Unix imports. It does not @@ -195,6 +199,7 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_ug_codex_app_server_client_initializes` | Connect a stdio client, direct/`--` separator, routing off/on | Actual JSON-RPC initialize response; no non-JSON stdout; no routing | | `test_smart_routing_claude_route_subagent_hook`, `test_smart_routing_codex_route_subagent_hook` | Pipe a real PreToolUse spawn payload to the installed route-subagent hook with subagent-only routing enabled | Allow decision against the live router; requested model replaced by a routed agent definition (Claude) or bundled catalog slug (Codex) from the offered models; one audited decision matching the session and task | | `test_smart_router_skill_toggles_claude_subagent_routing`, `test_smart_router_skill_toggles_codex_subagent_routing` | Configure, launch a real subagent-only TUI with orchestration unset or opted in, then spawn tagged children while invoking the installed Smart Router skill to switch routing on -> off -> on in the same session | Only `smart-router` is installed by default; opt-in also installs `orchestrate`; all three native children complete; only routing-enabled phases show the subagent banner and produce a live routing decision correlated with the child; no first-prompt routing wrapper; normal exit | +| `test_orchestrator_claude_delegates_and_recovers_after_compaction`, `test_orchestrator_codex_delegates_and_recovers_after_compaction` | Opt into orchestration, request a multi-file review without asking for subagents, compact natively, and request a different review | Both reviews have new completed child work and a root answer with the files' withheld review IDs; the full workflow is delivered before the first review and after native compaction | | `test_ug_configure_claude_repeat_and_revert`, `test_ug_configure_codex_repeat_and_revert` | Configure twice over user settings; complete a task; revert twice | Settings preserved; no bearer in ug state; generated config removed; status unconfigured | | `test_ug_configure_claude_cleans_stale_skills_mcp_on_workspace_switch` | Configure the first workspace, register its skills MCP, switch to a second real workspace, and use Claude | Old registration removed from Claude and the new workspace state; old workspace bucket preserved; repeat configure stays clean; real file task completes on the second workspace | | `test_ug_configure_claude_rejects_invalid_credentials`, `test_ug_configure_codex_rejects_invalid_credentials` | Configure with a rejected bearer against the real workspace | Authentication failure; no successful saved setup | diff --git a/tests/integration/README.md b/tests/integration/README.md index 73a2736cd..c5d2bccbd 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -226,6 +226,7 @@ test_ug_claude_commands.py # command help forwarding test_ug_codex_commands.py # command help and parser error forwarding test_ug_codex_app_server.py # actual client/server initialize exchange test_ug_smart_routing_hooks.py # live hook contract plus skill-driven subagent toggles +test_ug_orchestrator.py # automatic delegation and native compaction continuation test_ug_configure_claude_lifecycle.py # repeat setup, revert, rejected credentials test_ug_configure_claude_workspace_switch.py # real skills MCP cleanup across two workspaces test_ug_configure_codex_lifecycle.py # repeat setup, revert, rejected credentials @@ -289,11 +290,21 @@ assistant answer after each skill invocation. Collapsed terminal output is allowed; the answer need not repeat the CLI's exact wording. Each following child still verifies whether a routing decision occurred. Their off-phase child is an explicit user-requested delegation; -these journeys do not establish automatic orchestration behavior. Root-only -activation, compaction, retained skills in ineligible sessions, -role-contract preservation, and isolation from legacy preference files -lack dedicated regression coverage. Codex's native hook merging, project trust, -and execution of pre-existing hooks are not exercised by this integration suite. +these toggle journeys do not establish automatic orchestration behavior. + +`test_ug_orchestrator.py` covers automatic delegation separately in the managed-fixture +lane. Each real harness reviews three source modules without a request to use subagents, +then performs native `/compact` and reviews three different modules. Each review needs +new completed native child work and a root report containing the exact review IDs from +all source headers; the IDs are never included in the prompt. The test also requires +the full workflow in native hook context before the first review and after compaction. +It checks orchestration and continuation, not the accuracy of review findings. +Run these two journeys with `-- -m managed_fixture -k test_orchestrator_`. + +Retained skills in ineligible sessions, role-contract preservation, and isolation +from legacy preference files still lack dedicated regression coverage. Codex's native +hook merging, project trust, and execution of pre-existing hooks are not exercised by +this integration suite. The portable `../test_claude_windows_smart_routing.py` checks the Windows subagent-only fallback without Unix imports. Native Windows TUI and hook execution diff --git a/tests/integration/test_ug_orchestrator.py b/tests/integration/test_ug_orchestrator.py new file mode 100644 index 000000000..9daeaa2bd --- /dev/null +++ b/tests/integration/test_ug_orchestrator.py @@ -0,0 +1,206 @@ +"""Real automatic delegation and continuation after native context compaction.""" + +import uuid +from pathlib import Path + +import pytest +from utils.constants import CLAUDE_SMART_ROUTING_MODELS, CODEX_SMART_ROUTING_MODELS +from utils.evidence import ( + agent_sessions, + assert_no_terminal_api_error, + completed_answers, + completed_child_answers, + is_child_session, + orchestrator_contexts, +) +from utils.managed import ( + build_claude_agent_config, + build_codex_agent_config, + build_coding_agent_config, + set_managed_config_stub, +) +from utils.terminal import AgentTerminal + + +class ReviewTask: + """Substantive source review whose traceability values are absent from the prompt.""" + + def __init__(self, session, phase, sources): + directory = session.cwd / phase + directory.mkdir() + self.phase = phase + self.values = [] + source_root = Path(__file__).resolve().parents[2] / "src" / "ucode" + for source in sources: + value = uuid.uuid4().hex + self.values.append(value) + text = f"# REVIEW_ID: {value}\n" + (source_root / source).read_text() + (directory / Path(source).name).write_text(text) + self.prompt = ( + f"Review all three Python modules in {phase}/ for correctness, error handling, " + "and preservation of user data. Read each module and report concrete findings " + "with file references, or explain why you found none. These are standalone " + "review copies; dependencies outside this directory are not part of the review. " + "Do not modify files or execute the source. Include each module's exact " + "REVIEW_ID from its header in your final report so the review is traceable." + ) + + def completed(self, session, agent): + return any( + all(value in answer for value in self.values) + for answer in completed_answers(agent, _root_records(session, agent)) + ) + + +def _root_records(session, agent): + return [ + row + for path, records in agent_sessions(session, agent).items() + if not is_child_session(agent, path, records) + for row in records + ] + + +def _review(tui, session, agent, task): + before_children = completed_child_answers(session, agent) + tui.submit(task.prompt) + + def finished(screen): + assert_no_terminal_api_error(screen) + return task.completed(session, agent) + + tui.wait_for(finished, "a completed review of all three modules", timeout=360) + children = completed_child_answers(session, agent) + new_answers = { + path: answers[len(before_children.get(path, [])) :] for path, answers in children.items() + } + session.record( + f"{task.phase}-completion.json", + {"review_ids": task.values, "new_completed_child_answers": new_answers}, + ) + assert any(answer.strip() for answers in new_answers.values() for answer in answers), ( + "The root completed the review without a completed native child review" + ) + assert orchestrator_contexts(agent, _root_records(session, agent)), ( + "The native root transcript did not receive the orchestrator workflow" + ) + + +def _compact(tui, session, agent): + before = len(_root_records(session, agent)) + tui.submit("/compact") + + def reloaded(screen): + assert_no_terminal_api_error(screen) + records = _root_records(session, agent)[before:] + if agent == "claude": + compacted = any(row.get("subtype") == "compact_boundary" for row in records) + contexts = [ + row + for row in records + if row.get("attachment", {}).get("hookEvent") == "SessionStart" + ] + else: + compacted = any( + row.get("type") == "compacted" + or ( + row.get("type") == "event_msg" + and row.get("payload", {}).get("type") == "context_compacted" + ) + for row in records + ) + contexts = records + return compacted and bool(orchestrator_contexts(agent, contexts)) + + tui.wait_for(reloaded, "native compaction and a reloaded orchestrator workflow", timeout=180) + session.record("compaction-records.json", _root_records(session, agent)[before:]) + + +@pytest.mark.claude +@pytest.mark.managed_fixture +def test_orchestrator_claude_delegates_and_recovers_after_compaction( + live_session, workspace, tmp_path +): + """Scenario: opt into orchestration and ask Claude to review three source modules, + without requesting subagents; compact natively, then review three different modules. + + Expected: the prompt and compaction hooks deliver the complete workflow; both reviews + have a completed native child and a root report containing values read from all files. + This checks delegation and continuation, not the accuracy of the review's findings. + """ + session = live_session + session.env.update( + ENABLE_SMART_ROUTING_V2="1", + ENABLE_SMART_ROUTING_SUBAGENT_ONLY="1", + ENABLE_ORCHESTRATION="1", + TMPDIR=str(tmp_path), + ) + config = build_coding_agent_config( + "CODING_AGENT_CLAUDE_CODE", + build_claude_agent_config(CLAUDE_SMART_ROUTING_MODELS, smart_routing=True), + ) + set_managed_config_stub(session, tmp_path, config) + session.run( + "configure", "--workspace", workspace, "--skip-upgrade", "--disable-databricks-ai-tools" + ) + first = ReviewTask( + session, "initial", ["config_io.py", "gateway_proxy.py", "managed_config.py"] + ) + with AgentTerminal( + session, "claude", [str(session.binary), "claude"], "automatic-orchestration" + ) as tui: + tui.boot() + _review(tui, session, "claude", first) + _compact(tui, session, "claude") + second = ReviewTask( + session, + "followup", + ["skills_download.py", "codex_config.py", "smart_routing/session_env.py"], + ) + _review(tui, session, "claude", second) + tui.exit_normally() + + +@pytest.mark.codex +@pytest.mark.managed_fixture +def test_orchestrator_codex_delegates_and_recovers_after_compaction( + live_session, workspace, tmp_path +): + """Scenario: opt into orchestration and ask Codex to review three source modules, + without requesting subagents; compact natively, then review three different modules. + + Expected: the prompt and compaction hooks deliver the complete workflow; both reviews + have a completed native child and a root report containing values read from all files. + Parent history copied into a child is excluded from child-completion evidence. + """ + session = live_session + session.env.update( + ENABLE_SMART_ROUTING_V2="1", + ENABLE_SMART_ROUTING_SUBAGENT_ONLY="1", + ENABLE_ORCHESTRATION="1", + TMPDIR=str(tmp_path), + ) + config = build_coding_agent_config( + "CODING_AGENT_CODEX", + build_codex_agent_config(models=CODEX_SMART_ROUTING_MODELS, smart_routing=True), + ) + set_managed_config_stub(session, tmp_path, config) + session.run( + "configure", "--workspace", workspace, "--skip-upgrade", "--disable-databricks-ai-tools" + ) + first = ReviewTask( + session, "initial", ["config_io.py", "gateway_proxy.py", "managed_config.py"] + ) + with AgentTerminal( + session, "codex", [str(session.binary), "codex"], "automatic-orchestration" + ) as tui: + tui.boot() + _review(tui, session, "codex", first) + _compact(tui, session, "codex") + second = ReviewTask( + session, + "followup", + ["skills_download.py", "codex_config.py", "smart_routing/session_env.py"], + ) + _review(tui, session, "codex", second) + tui.exit_normally() diff --git a/tests/integration/utils/evidence.py b/tests/integration/utils/evidence.py index d3f77a893..fc13be8af 100644 --- a/tests/integration/utils/evidence.py +++ b/tests/integration/utils/evidence.py @@ -52,6 +52,58 @@ def assistant_answers(agent: str, records: list[dict]) -> list[str]: return helper.assistant_answers(records) if helper is not None else [] +def completed_answers(agent: str, records: list[dict]) -> list[str]: + """Exclude Claude's intermediate commentary from task-completion evidence.""" + if agent == "claude": + records = [ + row for row in records if row.get("message", {}).get("stop_reason") == "end_turn" + ] + return assistant_answers(agent, records) + + +def completed_child_answers(session, agent: str) -> dict[str, list[str]]: + """Do not count parent turns copied into a Codex child's rollout as child work.""" + sessions = agent_sessions(session, agent) + parent_turns = { + row["payload"]["turn_id"] + for path, records in sessions.items() + if not is_child_session(agent, path, records) + for row in records + if row.get("type") == "turn_context" and row.get("payload", {}).get("turn_id") + } + return { + path: completed_answers( + agent, + [row for row in records if row.get("payload", {}).get("turn_id") not in parent_turns], + ) + for path, records in sessions.items() + if is_child_session(agent, path, records) + } + + +def orchestrator_contexts(agent: str, records: list[dict]) -> list[str]: + """Read delivered hook context, excluding user text and assistant claims.""" + contexts = [] + for row in records: + if agent == "claude": + attachment = row.get("attachment", {}) + if attachment.get("type") == "hook_additional_context": + contexts.extend(attachment.get("content", [])) + elif row.get("type") == "response_item": + payload = row.get("payload", {}) + if payload.get("type") == "message" and payload.get("role") == "developer": + contexts.extend(part.get("text", "") for part in payload.get("content", [])) + return [ + text + for text in contexts + if isinstance(text, str) + and text.startswith("Smart Router Orchestrator is on for this session.") + and "## Workflow" in text + and "### Claude Code adapter" in text + and "### Codex adapter" in text + ] + + def tool_outputs(agent: str, records: list[dict]) -> list[str]: """Read native tool results even when their terminal output is collapsed.""" contents = [] diff --git a/tests/test_integration_evidence.py b/tests/test_integration_evidence.py index cf04efc76..7bf52e2bd 100644 --- a/tests/test_integration_evidence.py +++ b/tests/test_integration_evidence.py @@ -10,6 +10,9 @@ SubagentCalculation, assert_no_terminal_api_error, assistant_answer_contains, + completed_answers, + completed_child_answers, + orchestrator_contexts, tool_outputs, ) @@ -103,6 +106,88 @@ def test_tagged_calculation_requires_the_native_child_answer(tmp_path, agent): assert "1+1" in task.prompt +def test_completed_answers_excludes_claude_intermediate_commentary(): + records = [ + { + "type": "assistant", + "message": { + "role": "assistant", + "stop_reason": reason, + "content": [{"type": "text", "text": answer}], + }, + } + for reason, answer in [("tool_use", "I will inspect the module"), ("end_turn", "Reviewed")] + ] + assert completed_answers("claude", records) == ["Reviewed"] + + +def test_completed_child_answers_excludes_inherited_codex_parent_work(tmp_path): + parent = [ + {"type": "turn_context", "payload": {"turn_id": "parent-turn"}}, + { + "type": "event_msg", + "payload": { + "type": "task_complete", + "turn_id": "parent-turn", + "last_agent_message": "Parent answer", + }, + }, + ] + child = [ + {"type": "session_meta", "payload": {"source": {"subagent": "spawn"}}}, + *parent, + { + "type": "event_msg", + "payload": { + "type": "task_complete", + "turn_id": "child-turn", + "last_agent_message": "Child answer", + }, + }, + ] + session = _transcript_session(tmp_path, "codex", {"parent.jsonl": parent, "child.jsonl": child}) + assert completed_child_answers(session, "codex") == {"child.jsonl": ["Child answer"]} + + +@pytest.mark.parametrize("agent", ["claude", "codex"]) +def test_orchestrator_context_requires_native_delivery_not_echoed_text(agent): + context = ( + "Smart Router Orchestrator is on for this session.\n" + "## Workflow\n### Claude Code adapter\n### Codex adapter" + ) + records = [ + {"type": "user", "message": {"content": context}}, + { + "type": "response_item", + "payload": { + "type": "message", + "role": "user", + "content": [{"type": "input_text", "text": context}], + }, + }, + ] + assert orchestrator_contexts(agent, records) == [] + if agent == "claude": + records.append( + { + "type": "attachment", + "attachment": {"type": "hook_additional_context", "content": [context]}, + } + ) + else: + records.append( + { + "type": "response_item", + "payload": { + "type": "message", + "role": "developer", + "content": [{"type": "input_text", "text": context}], + }, + } + ) + assert orchestrator_contexts(agent, records) == [context] + + def test_codex_model_identity_uses_only_the_completed_answer_turn(): records = [ { From 78176dccb79d620dc54176acd92fe4a7ff71c3e8 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 22:20:37 +0000 Subject: [PATCH 20/25] Fix Codex compaction test timing --- tests/README.md | 5 +++-- tests/integration/README.md | 3 ++- tests/integration/test_ug_orchestrator.py | 11 +++++++---- 3 files changed, 12 insertions(+), 7 deletions(-) diff --git a/tests/README.md b/tests/README.md index 7d1a687d7..91fb324a0 100644 --- a/tests/README.md +++ b/tests/README.md @@ -142,8 +142,9 @@ including collapsed-output records, and excludes user echoes and assistant claim `test_ug_orchestrator.py` requests a review of three source modules without mentioning subagents. It requires a completed native child response and a root report containing traceability values read from all three files. Each harness then performs native -`/compact`, receives the full workflow again, and completes a second review with new -child work. The journeys check orchestration and continuation, not review accuracy. +`/compact`, receives the full workflow again for the next review, and completes it with +new child work. Codex delivers its compact hook before the next model request. +The journeys check orchestration and continuation, not review accuracy. Retained skills in ineligible sessions, role-contract preservation, and isolation from legacy preference files still lack dedicated regression coverage. Codex's native hook merging and project trust, and execution of pre-existing hooks remain diff --git a/tests/integration/README.md b/tests/integration/README.md index c5d2bccbd..09470f4a1 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -297,7 +297,8 @@ lane. Each real harness reviews three source modules without a request to use su then performs native `/compact` and reviews three different modules. Each review needs new completed native child work and a root report containing the exact review IDs from all source headers; the IDs are never included in the prompt. The test also requires -the full workflow in native hook context before the first review and after compaction. +the full workflow in fresh native hook context for both reviews. Codex runs its compact +hook before the next model request, so the second review checks that delivery. It checks orchestration and continuation, not the accuracy of review findings. Run these two journeys with `-- -m managed_fixture -k test_orchestrator_`. diff --git a/tests/integration/test_ug_orchestrator.py b/tests/integration/test_ug_orchestrator.py index 9daeaa2bd..5c431fe0b 100644 --- a/tests/integration/test_ug_orchestrator.py +++ b/tests/integration/test_ug_orchestrator.py @@ -63,6 +63,7 @@ def _root_records(session, agent): def _review(tui, session, agent, task): before_children = completed_child_answers(session, agent) + before_records = len(_root_records(session, agent)) tui.submit(task.prompt) def finished(screen): @@ -81,8 +82,8 @@ def finished(screen): assert any(answer.strip() for answers in new_answers.values() for answer in answers), ( "The root completed the review without a completed native child review" ) - assert orchestrator_contexts(agent, _root_records(session, agent)), ( - "The native root transcript did not receive the orchestrator workflow" + assert orchestrator_contexts(agent, _root_records(session, agent)[before_records:]), ( + "The native root transcript did not receive a fresh orchestrator workflow" ) @@ -109,10 +110,12 @@ def reloaded(screen): ) for row in records ) - contexts = records + # Codex runs the compact hook before the next model request. + # The follow-up review requires its newly delivered workflow. + return compacted return compacted and bool(orchestrator_contexts(agent, contexts)) - tui.wait_for(reloaded, "native compaction and a reloaded orchestrator workflow", timeout=180) + tui.wait_for(reloaded, "native compaction", timeout=180) session.record("compaction-records.json", _root_records(session, agent)[before:]) From 684aed20e8c961c88c576ed81759acec25fc78ed Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 22:24:51 +0000 Subject: [PATCH 21/25] Move temporary orchestration verification out of integration PR --- tests/README.md | 16 +- tests/integration/README.md | 22 +-- tests/integration/test_ug_orchestrator.py | 209 ---------------------- tests/integration/utils/evidence.py | 52 ------ tests/test_integration_evidence.py | 85 --------- 5 files changed, 10 insertions(+), 374 deletions(-) delete mode 100644 tests/integration/test_ug_orchestrator.py diff --git a/tests/README.md b/tests/README.md index 91fb324a0..d78d3e368 100644 --- a/tests/README.md +++ b/tests/README.md @@ -139,16 +139,11 @@ including while routing is off. `test_integration_evidence.py` checks native tool-result extraction for both agents, including collapsed-output records, and excludes user echoes and assistant claims. -`test_ug_orchestrator.py` requests a review of three source modules without mentioning -subagents. It requires a completed native child response and a root report containing -traceability values read from all three files. Each harness then performs native -`/compact`, receives the full workflow again for the next review, and completes it with -new child work. Codex delivers its compact hook before the next model request. -The journeys check orchestration and continuation, not review accuracy. -Retained skills in ineligible sessions, role-contract preservation, and isolation -from legacy preference files still lack dedicated regression coverage. Codex's -native hook merging and project trust, and execution of pre-existing hooks remain -outside the integration suite. +Dedicated regression coverage is missing for root-only orchestrator activation, +compaction, retained skills in ineligible sessions, +role-contract preservation, and isolation from legacy preference files. Codex's +native hook merging and project trust, automatic delegation, and execution of +pre-existing hooks remain outside the integration suite. The portable Windows routing test checks native executable forwarding, generated hooks/plugins, caller arguments, and cleanup without Unix imports. It does not @@ -200,7 +195,6 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_ug_codex_app_server_client_initializes` | Connect a stdio client, direct/`--` separator, routing off/on | Actual JSON-RPC initialize response; no non-JSON stdout; no routing | | `test_smart_routing_claude_route_subagent_hook`, `test_smart_routing_codex_route_subagent_hook` | Pipe a real PreToolUse spawn payload to the installed route-subagent hook with subagent-only routing enabled | Allow decision against the live router; requested model replaced by a routed agent definition (Claude) or bundled catalog slug (Codex) from the offered models; one audited decision matching the session and task | | `test_smart_router_skill_toggles_claude_subagent_routing`, `test_smart_router_skill_toggles_codex_subagent_routing` | Configure, launch a real subagent-only TUI with orchestration unset or opted in, then spawn tagged children while invoking the installed Smart Router skill to switch routing on -> off -> on in the same session | Only `smart-router` is installed by default; opt-in also installs `orchestrate`; all three native children complete; only routing-enabled phases show the subagent banner and produce a live routing decision correlated with the child; no first-prompt routing wrapper; normal exit | -| `test_orchestrator_claude_delegates_and_recovers_after_compaction`, `test_orchestrator_codex_delegates_and_recovers_after_compaction` | Opt into orchestration, request a multi-file review without asking for subagents, compact natively, and request a different review | Both reviews have new completed child work and a root answer with the files' withheld review IDs; the full workflow is delivered before the first review and after native compaction | | `test_ug_configure_claude_repeat_and_revert`, `test_ug_configure_codex_repeat_and_revert` | Configure twice over user settings; complete a task; revert twice | Settings preserved; no bearer in ug state; generated config removed; status unconfigured | | `test_ug_configure_claude_cleans_stale_skills_mcp_on_workspace_switch` | Configure the first workspace, register its skills MCP, switch to a second real workspace, and use Claude | Old registration removed from Claude and the new workspace state; old workspace bucket preserved; repeat configure stays clean; real file task completes on the second workspace | | `test_ug_configure_claude_rejects_invalid_credentials`, `test_ug_configure_codex_rejects_invalid_credentials` | Configure with a rejected bearer against the real workspace | Authentication failure; no successful saved setup | diff --git a/tests/integration/README.md b/tests/integration/README.md index 09470f4a1..73a2736cd 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -226,7 +226,6 @@ test_ug_claude_commands.py # command help forwarding test_ug_codex_commands.py # command help and parser error forwarding test_ug_codex_app_server.py # actual client/server initialize exchange test_ug_smart_routing_hooks.py # live hook contract plus skill-driven subagent toggles -test_ug_orchestrator.py # automatic delegation and native compaction continuation test_ug_configure_claude_lifecycle.py # repeat setup, revert, rejected credentials test_ug_configure_claude_workspace_switch.py # real skills MCP cleanup across two workspaces test_ug_configure_codex_lifecycle.py # repeat setup, revert, rejected credentials @@ -290,22 +289,11 @@ assistant answer after each skill invocation. Collapsed terminal output is allowed; the answer need not repeat the CLI's exact wording. Each following child still verifies whether a routing decision occurred. Their off-phase child is an explicit user-requested delegation; -these toggle journeys do not establish automatic orchestration behavior. - -`test_ug_orchestrator.py` covers automatic delegation separately in the managed-fixture -lane. Each real harness reviews three source modules without a request to use subagents, -then performs native `/compact` and reviews three different modules. Each review needs -new completed native child work and a root report containing the exact review IDs from -all source headers; the IDs are never included in the prompt. The test also requires -the full workflow in fresh native hook context for both reviews. Codex runs its compact -hook before the next model request, so the second review checks that delivery. -It checks orchestration and continuation, not the accuracy of review findings. -Run these two journeys with `-- -m managed_fixture -k test_orchestrator_`. - -Retained skills in ineligible sessions, role-contract preservation, and isolation -from legacy preference files still lack dedicated regression coverage. Codex's native -hook merging, project trust, and execution of pre-existing hooks are not exercised by -this integration suite. +these journeys do not establish automatic orchestration behavior. Root-only +activation, compaction, retained skills in ineligible sessions, +role-contract preservation, and isolation from legacy preference files +lack dedicated regression coverage. Codex's native hook merging, project trust, +and execution of pre-existing hooks are not exercised by this integration suite. The portable `../test_claude_windows_smart_routing.py` checks the Windows subagent-only fallback without Unix imports. Native Windows TUI and hook execution diff --git a/tests/integration/test_ug_orchestrator.py b/tests/integration/test_ug_orchestrator.py deleted file mode 100644 index 5c431fe0b..000000000 --- a/tests/integration/test_ug_orchestrator.py +++ /dev/null @@ -1,209 +0,0 @@ -"""Real automatic delegation and continuation after native context compaction.""" - -import uuid -from pathlib import Path - -import pytest -from utils.constants import CLAUDE_SMART_ROUTING_MODELS, CODEX_SMART_ROUTING_MODELS -from utils.evidence import ( - agent_sessions, - assert_no_terminal_api_error, - completed_answers, - completed_child_answers, - is_child_session, - orchestrator_contexts, -) -from utils.managed import ( - build_claude_agent_config, - build_codex_agent_config, - build_coding_agent_config, - set_managed_config_stub, -) -from utils.terminal import AgentTerminal - - -class ReviewTask: - """Substantive source review whose traceability values are absent from the prompt.""" - - def __init__(self, session, phase, sources): - directory = session.cwd / phase - directory.mkdir() - self.phase = phase - self.values = [] - source_root = Path(__file__).resolve().parents[2] / "src" / "ucode" - for source in sources: - value = uuid.uuid4().hex - self.values.append(value) - text = f"# REVIEW_ID: {value}\n" + (source_root / source).read_text() - (directory / Path(source).name).write_text(text) - self.prompt = ( - f"Review all three Python modules in {phase}/ for correctness, error handling, " - "and preservation of user data. Read each module and report concrete findings " - "with file references, or explain why you found none. These are standalone " - "review copies; dependencies outside this directory are not part of the review. " - "Do not modify files or execute the source. Include each module's exact " - "REVIEW_ID from its header in your final report so the review is traceable." - ) - - def completed(self, session, agent): - return any( - all(value in answer for value in self.values) - for answer in completed_answers(agent, _root_records(session, agent)) - ) - - -def _root_records(session, agent): - return [ - row - for path, records in agent_sessions(session, agent).items() - if not is_child_session(agent, path, records) - for row in records - ] - - -def _review(tui, session, agent, task): - before_children = completed_child_answers(session, agent) - before_records = len(_root_records(session, agent)) - tui.submit(task.prompt) - - def finished(screen): - assert_no_terminal_api_error(screen) - return task.completed(session, agent) - - tui.wait_for(finished, "a completed review of all three modules", timeout=360) - children = completed_child_answers(session, agent) - new_answers = { - path: answers[len(before_children.get(path, [])) :] for path, answers in children.items() - } - session.record( - f"{task.phase}-completion.json", - {"review_ids": task.values, "new_completed_child_answers": new_answers}, - ) - assert any(answer.strip() for answers in new_answers.values() for answer in answers), ( - "The root completed the review without a completed native child review" - ) - assert orchestrator_contexts(agent, _root_records(session, agent)[before_records:]), ( - "The native root transcript did not receive a fresh orchestrator workflow" - ) - - -def _compact(tui, session, agent): - before = len(_root_records(session, agent)) - tui.submit("/compact") - - def reloaded(screen): - assert_no_terminal_api_error(screen) - records = _root_records(session, agent)[before:] - if agent == "claude": - compacted = any(row.get("subtype") == "compact_boundary" for row in records) - contexts = [ - row - for row in records - if row.get("attachment", {}).get("hookEvent") == "SessionStart" - ] - else: - compacted = any( - row.get("type") == "compacted" - or ( - row.get("type") == "event_msg" - and row.get("payload", {}).get("type") == "context_compacted" - ) - for row in records - ) - # Codex runs the compact hook before the next model request. - # The follow-up review requires its newly delivered workflow. - return compacted - return compacted and bool(orchestrator_contexts(agent, contexts)) - - tui.wait_for(reloaded, "native compaction", timeout=180) - session.record("compaction-records.json", _root_records(session, agent)[before:]) - - -@pytest.mark.claude -@pytest.mark.managed_fixture -def test_orchestrator_claude_delegates_and_recovers_after_compaction( - live_session, workspace, tmp_path -): - """Scenario: opt into orchestration and ask Claude to review three source modules, - without requesting subagents; compact natively, then review three different modules. - - Expected: the prompt and compaction hooks deliver the complete workflow; both reviews - have a completed native child and a root report containing values read from all files. - This checks delegation and continuation, not the accuracy of the review's findings. - """ - session = live_session - session.env.update( - ENABLE_SMART_ROUTING_V2="1", - ENABLE_SMART_ROUTING_SUBAGENT_ONLY="1", - ENABLE_ORCHESTRATION="1", - TMPDIR=str(tmp_path), - ) - config = build_coding_agent_config( - "CODING_AGENT_CLAUDE_CODE", - build_claude_agent_config(CLAUDE_SMART_ROUTING_MODELS, smart_routing=True), - ) - set_managed_config_stub(session, tmp_path, config) - session.run( - "configure", "--workspace", workspace, "--skip-upgrade", "--disable-databricks-ai-tools" - ) - first = ReviewTask( - session, "initial", ["config_io.py", "gateway_proxy.py", "managed_config.py"] - ) - with AgentTerminal( - session, "claude", [str(session.binary), "claude"], "automatic-orchestration" - ) as tui: - tui.boot() - _review(tui, session, "claude", first) - _compact(tui, session, "claude") - second = ReviewTask( - session, - "followup", - ["skills_download.py", "codex_config.py", "smart_routing/session_env.py"], - ) - _review(tui, session, "claude", second) - tui.exit_normally() - - -@pytest.mark.codex -@pytest.mark.managed_fixture -def test_orchestrator_codex_delegates_and_recovers_after_compaction( - live_session, workspace, tmp_path -): - """Scenario: opt into orchestration and ask Codex to review three source modules, - without requesting subagents; compact natively, then review three different modules. - - Expected: the prompt and compaction hooks deliver the complete workflow; both reviews - have a completed native child and a root report containing values read from all files. - Parent history copied into a child is excluded from child-completion evidence. - """ - session = live_session - session.env.update( - ENABLE_SMART_ROUTING_V2="1", - ENABLE_SMART_ROUTING_SUBAGENT_ONLY="1", - ENABLE_ORCHESTRATION="1", - TMPDIR=str(tmp_path), - ) - config = build_coding_agent_config( - "CODING_AGENT_CODEX", - build_codex_agent_config(models=CODEX_SMART_ROUTING_MODELS, smart_routing=True), - ) - set_managed_config_stub(session, tmp_path, config) - session.run( - "configure", "--workspace", workspace, "--skip-upgrade", "--disable-databricks-ai-tools" - ) - first = ReviewTask( - session, "initial", ["config_io.py", "gateway_proxy.py", "managed_config.py"] - ) - with AgentTerminal( - session, "codex", [str(session.binary), "codex"], "automatic-orchestration" - ) as tui: - tui.boot() - _review(tui, session, "codex", first) - _compact(tui, session, "codex") - second = ReviewTask( - session, - "followup", - ["skills_download.py", "codex_config.py", "smart_routing/session_env.py"], - ) - _review(tui, session, "codex", second) - tui.exit_normally() diff --git a/tests/integration/utils/evidence.py b/tests/integration/utils/evidence.py index fc13be8af..d3f77a893 100644 --- a/tests/integration/utils/evidence.py +++ b/tests/integration/utils/evidence.py @@ -52,58 +52,6 @@ def assistant_answers(agent: str, records: list[dict]) -> list[str]: return helper.assistant_answers(records) if helper is not None else [] -def completed_answers(agent: str, records: list[dict]) -> list[str]: - """Exclude Claude's intermediate commentary from task-completion evidence.""" - if agent == "claude": - records = [ - row for row in records if row.get("message", {}).get("stop_reason") == "end_turn" - ] - return assistant_answers(agent, records) - - -def completed_child_answers(session, agent: str) -> dict[str, list[str]]: - """Do not count parent turns copied into a Codex child's rollout as child work.""" - sessions = agent_sessions(session, agent) - parent_turns = { - row["payload"]["turn_id"] - for path, records in sessions.items() - if not is_child_session(agent, path, records) - for row in records - if row.get("type") == "turn_context" and row.get("payload", {}).get("turn_id") - } - return { - path: completed_answers( - agent, - [row for row in records if row.get("payload", {}).get("turn_id") not in parent_turns], - ) - for path, records in sessions.items() - if is_child_session(agent, path, records) - } - - -def orchestrator_contexts(agent: str, records: list[dict]) -> list[str]: - """Read delivered hook context, excluding user text and assistant claims.""" - contexts = [] - for row in records: - if agent == "claude": - attachment = row.get("attachment", {}) - if attachment.get("type") == "hook_additional_context": - contexts.extend(attachment.get("content", [])) - elif row.get("type") == "response_item": - payload = row.get("payload", {}) - if payload.get("type") == "message" and payload.get("role") == "developer": - contexts.extend(part.get("text", "") for part in payload.get("content", [])) - return [ - text - for text in contexts - if isinstance(text, str) - and text.startswith("Smart Router Orchestrator is on for this session.") - and "## Workflow" in text - and "### Claude Code adapter" in text - and "### Codex adapter" in text - ] - - def tool_outputs(agent: str, records: list[dict]) -> list[str]: """Read native tool results even when their terminal output is collapsed.""" contents = [] diff --git a/tests/test_integration_evidence.py b/tests/test_integration_evidence.py index 7bf52e2bd..cf04efc76 100644 --- a/tests/test_integration_evidence.py +++ b/tests/test_integration_evidence.py @@ -10,9 +10,6 @@ SubagentCalculation, assert_no_terminal_api_error, assistant_answer_contains, - completed_answers, - completed_child_answers, - orchestrator_contexts, tool_outputs, ) @@ -106,88 +103,6 @@ def test_tagged_calculation_requires_the_native_child_answer(tmp_path, agent): assert "1+1" in task.prompt -def test_completed_answers_excludes_claude_intermediate_commentary(): - records = [ - { - "type": "assistant", - "message": { - "role": "assistant", - "stop_reason": reason, - "content": [{"type": "text", "text": answer}], - }, - } - for reason, answer in [("tool_use", "I will inspect the module"), ("end_turn", "Reviewed")] - ] - assert completed_answers("claude", records) == ["Reviewed"] - - -def test_completed_child_answers_excludes_inherited_codex_parent_work(tmp_path): - parent = [ - {"type": "turn_context", "payload": {"turn_id": "parent-turn"}}, - { - "type": "event_msg", - "payload": { - "type": "task_complete", - "turn_id": "parent-turn", - "last_agent_message": "Parent answer", - }, - }, - ] - child = [ - {"type": "session_meta", "payload": {"source": {"subagent": "spawn"}}}, - *parent, - { - "type": "event_msg", - "payload": { - "type": "task_complete", - "turn_id": "child-turn", - "last_agent_message": "Child answer", - }, - }, - ] - session = _transcript_session(tmp_path, "codex", {"parent.jsonl": parent, "child.jsonl": child}) - assert completed_child_answers(session, "codex") == {"child.jsonl": ["Child answer"]} - - -@pytest.mark.parametrize("agent", ["claude", "codex"]) -def test_orchestrator_context_requires_native_delivery_not_echoed_text(agent): - context = ( - "Smart Router Orchestrator is on for this session.\n" - "## Workflow\n### Claude Code adapter\n### Codex adapter" - ) - records = [ - {"type": "user", "message": {"content": context}}, - { - "type": "response_item", - "payload": { - "type": "message", - "role": "user", - "content": [{"type": "input_text", "text": context}], - }, - }, - ] - assert orchestrator_contexts(agent, records) == [] - if agent == "claude": - records.append( - { - "type": "attachment", - "attachment": {"type": "hook_additional_context", "content": [context]}, - } - ) - else: - records.append( - { - "type": "response_item", - "payload": { - "type": "message", - "role": "developer", - "content": [{"type": "input_text", "text": context}], - }, - } - ) - assert orchestrator_contexts(agent, records) == [context] - - def test_codex_model_identity_uses_only_the_completed_answer_turn(): records = [ { From a098d13a48773fbad5ba3e73e52b305abdc34ba1 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Wed, 7 Oct 2026 23:05:38 +0000 Subject: [PATCH 22/25] Rename Smart Router Orchestrator identifiers --- AGENTS.md | 4 ++-- README.md | 12 ++++++------ .../LICENSE.upstream | 0 .../README.md | 6 +++--- .../SKILL.md | 10 +++++----- .../agents/explorer.md | 0 .../agents/researcher.md | 0 .../agents/reviewer.md | 0 .../agents/tester.md | 0 .../agents/worker.md | 0 skills/smart-router/SKILL.md | 12 ++++++------ src/ucode/constants.py | 2 +- src/ucode/skills.py | 2 +- src/ucode/smart_routing/orchestrator.py | 10 +++++----- src/ucode/smart_routing/v2.py | 8 ++++---- tests/README.md | 8 ++++---- tests/integration/README.md | 6 +++--- tests/integration/test_ug_smart_routing_hooks.py | 16 ++++++++-------- 18 files changed, 48 insertions(+), 48 deletions(-) rename skills/{orchestrate => smart-router-orchestrator}/LICENSE.upstream (100%) rename skills/{orchestrate => smart-router-orchestrator}/README.md (90%) rename skills/{orchestrate => smart-router-orchestrator}/SKILL.md (93%) rename skills/{orchestrate => smart-router-orchestrator}/agents/explorer.md (100%) rename skills/{orchestrate => smart-router-orchestrator}/agents/researcher.md (100%) rename skills/{orchestrate => smart-router-orchestrator}/agents/reviewer.md (100%) rename skills/{orchestrate => smart-router-orchestrator}/agents/tester.md (100%) rename skills/{orchestrate => smart-router-orchestrator}/agents/worker.md (100%) diff --git a/AGENTS.md b/AGENTS.md index 48b9b4053..156609e5f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -70,7 +70,7 @@ Fields live in `~/.claude/ucode-settings.json` and the OS-managed settings file | Tracing | Ignore | Create/replace | The seven `CLAUDE_CODE_*`/`OTEL_*` trace keys and `otelHeadersHelper`; only when the config enables tracing | | `managedMcpServers` | Ignore | Merge | Add/update the config's MCP server entries; other entries left alone | | Smart-routing hooks | Merge | Merge | `PreToolUse`, `SessionStart`, `SubagentStart`; only `ug`'s own marked handlers, other hooks left alone | -| Orchestrator hooks | Merge | Merge | Launch-only `UserPromptSubmit` and compact `SessionStart` handlers for smart-routed sessions with `ENABLE_ORCHESTRATION=1`; read the same session controls as routing | +| Smart Router Orchestrator hooks | Merge | Merge | Launch-only `UserPromptSubmit` and compact `SessionStart` handlers for smart-routed sessions with `ENABLE_SMART_ROUTER_ORCHESTRATOR=1`; read the same session controls as routing | @@ -87,6 +87,6 @@ Fields live in `~/.codex/ucode.config.toml` and `/etc/codex/managed_config.toml` | `http_headers` | Merge | Merge | In `[model_providers.Databricks]`; merge `ug`'s routing headers by name, admin headers added under managed config | | `model_catalog_json` | Create/replace | Create/replace | In `~/.codex/config.toml`; `ug`'s own catalog reference, for a static model list | | `mcp_servers` | Ignore | Merge | Managed file; add/update the config's MCP server entries, other entries left alone | -| Smart-routing and orchestrator hooks | Merge | Merge | Launch-only `PreToolUse`, plus `UserPromptSubmit` and compact `SessionStart` with `ENABLE_ORCHESTRATION=1`; Codex combines them with its native hook sources and applies project trust; `features.hooks` is enabled for a smart-routed launch | +| Smart Router and Smart Router Orchestrator hooks | Merge | Merge | Launch-only `PreToolUse`, plus `UserPromptSubmit` and compact `SessionStart` with `ENABLE_SMART_ROUTER_ORCHESTRATOR=1`; Codex combines them with its native hook sources and applies project trust; `features.hooks` is enabled for a smart-routed launch | diff --git a/README.md b/README.md index 9fc527ec9..2a93de5ec 100644 --- a/README.md +++ b/README.md @@ -249,15 +249,15 @@ The generated shell hooks expect Git Bash; PowerShell-only setups are not covere ### Smart Router Orchestrator Smart-routed Claude and Codex sessions install `smart-router`. Set -`ENABLE_ORCHESTRATION=1` at launch to also install and activate Smart Router -Orchestrator through the bundled `orchestrate` skill; orchestration is off by +`ENABLE_SMART_ROUTER_ORCHESTRATOR=1` at launch to also install and activate Smart Router +Orchestrator through the bundled `smart-router-orchestrator` skill; orchestration is off by default. For example: ```bash -ENABLE_ORCHESTRATION=1 ENABLE_SMART_ROUTING_SUBAGENT_ONLY=1 ug claude +ENABLE_SMART_ROUTER_ORCHESTRATOR=1 ENABLE_SMART_ROUTING_SUBAGENT_ONLY=1 ug claude ``` -Use `ug codex` in the same command for Codex. The orchestrator assigns bounded work +Use `ug codex` in the same command for Codex. Smart Router Orchestrator assigns bounded work to explorer, researcher, worker, tester, and reviewer roles while the root plans, integrates, and verifies results. Easy tasks and explicit requests not to delegate stay in the root. @@ -267,7 +267,7 @@ and session controls. Turning Smart Router off through its skill stops new autom turning it on restores orchestration only in opted-in sessions. Explicit user requests for subagents still use normal harness behavior while routing is off. Stored skill files do not activate orchestration when the feature flag is unset or -`ENABLE_ORCHESTRATION=0`, or in non-routed sessions. Existing Isaac pilot gating +`ENABLE_SMART_ROUTER_ORCHESTRATOR=0`, or in non-routed sessions. Existing Isaac pilot gating and UG launch exclusions still apply. Hooks refresh orchestration state before each prompt and after compaction. A @@ -277,7 +277,7 @@ routing still checks the controls for each subagent. UG supplies its own hooks; Codex combines them with existing hooks and applies project trust. Smart routing selects subagent models; separate role-model preferences are ignored and their files are left untouched. See the bundled -[Smart Router Orchestrator documentation](skills/orchestrate/README.md) for details. +[Smart Router Orchestrator documentation](skills/smart-router-orchestrator/README.md) for details. ## Managed Files diff --git a/skills/orchestrate/LICENSE.upstream b/skills/smart-router-orchestrator/LICENSE.upstream similarity index 100% rename from skills/orchestrate/LICENSE.upstream rename to skills/smart-router-orchestrator/LICENSE.upstream diff --git a/skills/orchestrate/README.md b/skills/smart-router-orchestrator/README.md similarity index 90% rename from skills/orchestrate/README.md rename to skills/smart-router-orchestrator/README.md index 7a31e0823..9e5aad4ed 100644 --- a/skills/orchestrate/README.md +++ b/skills/smart-router-orchestrator/README.md @@ -1,8 +1,8 @@ # Smart Router Orchestrator -UG bundles the `orchestrate` workflow and five Claude role definitions. +UG bundles the `smart-router-orchestrator` workflow and five Claude role definitions. Smart-routed Claude and Codex launches install and -activate this skill alongside `smart-router` only with `ENABLE_ORCHESTRATION=1`. +activate this skill alongside `smart-router` only with `ENABLE_SMART_ROUTER_ORCHESTRATOR=1`. The feature is off by default; routing alone installs only `smart-router`. The workflow is injected before root prompts and after compaction. The hook checks @@ -14,7 +14,7 @@ opted-in sessions. A change made outside the conversation is observed at the nex prompt or compaction; model routing still checks the controls for each subagent. An installed skill or saved model preference cannot enable orchestration. Explicit user requests for subagents still use native harness behavior while routing -is off, without the orchestrator's workflow. +is off, without the Smart Router Orchestrator workflow. User instructions take precedence, and easy tasks remain in the root. Claude loads the bundled roles as `ug-smart-router:` in its temporary diff --git a/skills/orchestrate/SKILL.md b/skills/smart-router-orchestrator/SKILL.md similarity index 93% rename from skills/orchestrate/SKILL.md rename to skills/smart-router-orchestrator/SKILL.md index 7143fc9d4..21787f23b 100644 --- a/skills/orchestrate/SKILL.md +++ b/skills/smart-router-orchestrator/SKILL.md @@ -1,6 +1,6 @@ --- -name: orchestrate -description: Smart Router Orchestrator coordinates substantive development with native subagents when ENABLE_ORCHESTRATION=1 and Unity Gateway smart routing is enabled. Follow UG's activation context. Skip easy tasks and explicit no-subagent requests. +name: smart-router-orchestrator +description: Smart Router Orchestrator coordinates substantive development with native subagents when ENABLE_SMART_ROUTER_ORCHESTRATOR=1 and Unity Gateway smart routing is enabled. Follow UG's activation context. Skip easy tasks and explicit no-subagent requests. model: inherit argument-hint: "[task]" metadata: @@ -12,7 +12,7 @@ metadata: ## Activation UG's prompt and compaction hooks activate this workflow only in an eligible -smart-routing session launched with `ENABLE_ORCHESTRATION=1` while routing is on. +smart-routing session launched with `ENABLE_SMART_ROUTER_ORCHESTRATOR=1` while routing is on. Follow the latest UG activation context and successful Smart Router toggles; installed skill files and old context do not enable it. Use that context without running a separate pre-delegation check. Do not set flags or create a session to @@ -20,11 +20,11 @@ activate this workflow. Turning Smart Router off through its skill stops this workflow and supersedes earlier orchestration instructions. Do not start new automatic delegation or use -orchestrator role models as a fallback. Continue in the root unless the user +Smart Router Orchestrator role models as a fallback. Continue in the root unless the user explicitly requests a subagent; honor that request using the native tool and normal harness model selection, without this workflow. Keep routing off and collect results from existing children. Turning Smart Router back on restores -this workflow only if the session was launched with `ENABLE_ORCHESTRATION=1`. +this workflow only if the session was launched with `ENABLE_SMART_ROUTER_ORCHESTRATOR=1`. Use the `smart-router` skill only when the user asks to change routing. ## Workflow diff --git a/skills/orchestrate/agents/explorer.md b/skills/smart-router-orchestrator/agents/explorer.md similarity index 100% rename from skills/orchestrate/agents/explorer.md rename to skills/smart-router-orchestrator/agents/explorer.md diff --git a/skills/orchestrate/agents/researcher.md b/skills/smart-router-orchestrator/agents/researcher.md similarity index 100% rename from skills/orchestrate/agents/researcher.md rename to skills/smart-router-orchestrator/agents/researcher.md diff --git a/skills/orchestrate/agents/reviewer.md b/skills/smart-router-orchestrator/agents/reviewer.md similarity index 100% rename from skills/orchestrate/agents/reviewer.md rename to skills/smart-router-orchestrator/agents/reviewer.md diff --git a/skills/orchestrate/agents/tester.md b/skills/smart-router-orchestrator/agents/tester.md similarity index 100% rename from skills/orchestrate/agents/tester.md rename to skills/smart-router-orchestrator/agents/tester.md diff --git a/skills/orchestrate/agents/worker.md b/skills/smart-router-orchestrator/agents/worker.md similarity index 100% rename from skills/orchestrate/agents/worker.md rename to skills/smart-router-orchestrator/agents/worker.md diff --git a/skills/smart-router/SKILL.md b/skills/smart-router/SKILL.md index ace674eed..439f8d4a8 100644 --- a/skills/smart-router/SKILL.md +++ b/skills/smart-router/SKILL.md @@ -23,11 +23,11 @@ to restart through an updated Unity Gateway with smart routing enabled. With no argument, explain that only `on` and `off` are accepted. Do not edit the state file. This affects subsequent subagent model selection in the current session, not the root model -or first prompt. In sessions launched with `ENABLE_ORCHESTRATION=1`, it also controls Smart Router -Orchestrator. When turned off, earlier orchestrate instructions -are superseded: do not start new automatic delegation or fall back to orchestrator role models. +or first prompt. In sessions launched with `ENABLE_SMART_ROUTER_ORCHESTRATOR=1`, it also controls Smart Router +Orchestrator. When turned off, earlier Smart Router Orchestrator instructions +are superseded: do not start new automatic delegation or fall back to Smart Router Orchestrator role models. Continue in the root unless the user explicitly requests a subagent. Honor that request using -native tools and normal harness model selection, without the orchestrator; keep routing off. -Existing children can finish. When turned on, apply the orchestrate -skill to further work only if the session was launched with `ENABLE_ORCHESTRATION=1`. +native tools and normal harness model selection, without Smart Router Orchestrator; keep routing off. +Existing children can finish. When turned on, apply the `smart-router-orchestrator` +skill to further work only if the session was launched with `ENABLE_SMART_ROUTER_ORCHESTRATOR=1`. Do not change the orchestration feature flag. Return the command's result. diff --git a/src/ucode/constants.py b/src/ucode/constants.py index 25b847964..38fe7e774 100644 --- a/src/ucode/constants.py +++ b/src/ucode/constants.py @@ -5,7 +5,7 @@ ENABLE_SMART_ROUTING_ENV_VAR = "ENABLE_SMART_ROUTING_V2" ENABLE_SUBAGENT_ROUTING_ENV_VAR = "ENABLE_SMART_ROUTING_SUBAGENT_ONLY" -ENABLE_ORCHESTRATION_ENV_VAR = "ENABLE_ORCHESTRATION" +ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR = "ENABLE_SMART_ROUTER_ORCHESTRATOR" SMART_ROUTING_ENV_KEYS = ( ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, diff --git a/src/ucode/skills.py b/src/ucode/skills.py index 26318c270..0adb49688 100644 --- a/src/ucode/skills.py +++ b/src/ucode/skills.py @@ -13,7 +13,7 @@ _SKILL_NAME_PATTERN = re.compile(r"[a-z0-9]+(?:-[a-z0-9]+)*") SMART_ROUTER_SKILL = "smart-router" # Install the delegation workflow in the same eligible sessions as Smart Router. -ORCHESTRATOR_SKILL = "orchestrate" +SMART_ROUTER_ORCHESTRATOR_SKILL = "smart-router-orchestrator" def _skills_source() -> Path: diff --git a/src/ucode/smart_routing/orchestrator.py b/src/ucode/smart_routing/orchestrator.py index 937774f67..c86d4b0a5 100644 --- a/src/ucode/smart_routing/orchestrator.py +++ b/src/ucode/smart_routing/orchestrator.py @@ -13,22 +13,22 @@ from pathlib import Path from ucode import skills -from ucode.constants import ENABLE_ORCHESTRATION_ENV_VAR +from ucode.constants import ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR from ucode.smart_routing.hooks import sync_managed_hooks from ucode.smart_routing.session_env import effective_environment, session_env_path HOOK_MODULE = "ucode.smart_routing.orchestrator" DISABLED_CONTEXT = ( "Smart Router Orchestrator is off for this session. This supersedes earlier " - "orchestrator instructions. Continue in the root unless the user explicitly requests " - "subagents; use native tools and the current Smart Router setting for those requests. " + "Smart Router Orchestrator instructions. Continue in the root unless the user explicitly " + "requests subagents; use native tools and the current Smart Router setting for those requests. " "Collect results from children already running." ) def feature_enabled(env: Mapping[str, str] | None = None) -> bool: source = os.environ if env is None else env - return source.get(ENABLE_ORCHESTRATION_ENV_VAR) == "1" + return source.get(ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR) == "1" def enabled(env: Mapping[str, str] | None = None) -> bool: @@ -49,7 +49,7 @@ def enabled(env: Mapping[str, str] | None = None) -> bool: def skill_directory() -> Path: - return skills._skills_source() / skills.ORCHESTRATOR_SKILL + return skills._skills_source() / skills.SMART_ROUTER_ORCHESTRATOR_SKILL def add_claude_agents(plugin_dir: Path) -> None: diff --git a/src/ucode/smart_routing/v2.py b/src/ucode/smart_routing/v2.py index 8db8a1bff..0cd9205cf 100644 --- a/src/ucode/smart_routing/v2.py +++ b/src/ucode/smart_routing/v2.py @@ -27,7 +27,7 @@ write_text_file, ) from ucode.constants import ( - ENABLE_ORCHESTRATION_ENV_VAR, + ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR, ENABLE_SMART_ROUTING_ENV_VAR, ENABLE_SUBAGENT_ROUTING_ENV_VAR, LOOPBACK_HOST, @@ -47,7 +47,7 @@ acquire_exclusive_file_lock, release_file_lock, ) -from ucode.skills import ORCHESTRATOR_SKILL, SMART_ROUTER_SKILL, install_skill +from ucode.skills import SMART_ROUTER_ORCHESTRATOR_SKILL, SMART_ROUTER_SKILL, install_skill from ucode.smart_routing import claude_routing, codex_interposer, orchestrator, routing from ucode.smart_routing.claude_hooks import ( FIRST_PROMPT_SOCKET_ENV, @@ -86,7 +86,7 @@ class ClaudeRoutingSetupError(RuntimeError): def _prepare_smart_router_session(agent: str) -> Path: skills = [SMART_ROUTER_SKILL] if orchestrator.feature_enabled(): - skills.append(ORCHESTRATOR_SKILL) + skills.append(SMART_ROUTER_ORCHESTRATOR_SKILL) for skill in skills: try: install_skill(skill, agent, config_io.APP_DIR.parent) @@ -528,7 +528,7 @@ def launch_claude( if not isinstance(env, dict): raise RuntimeError("Claude settings 'env' must be an object for smart routing.") env.pop("CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY", None) - env[ENABLE_ORCHESTRATION_ENV_VAR] = "1" if orchestrator.feature_enabled() else "0" + env[ENABLE_SMART_ROUTER_ORCHESTRATOR_ENV_VAR] = "1" if orchestrator.feature_enabled() else "0" if route_first_prompt: env[ENABLE_SMART_ROUTING_ENV_VAR] = "1" else: diff --git a/tests/README.md b/tests/README.md index d78d3e368..72ebf5cc0 100644 --- a/tests/README.md +++ b/tests/README.md @@ -131,15 +131,15 @@ that Claude settings and Codex's shell policy carry the interpreter and session These are component checks; they do not establish native skill permission matching or PowerShell execution. -The toggle integration journeys run with `ENABLE_ORCHESTRATION` unset and with -`ENABLE_ORCHESTRATION=1`. They require only `smart-router` by default and both +The toggle integration journeys run with `ENABLE_SMART_ROUTER_ORCHESTRATOR` unset and with +`ENABLE_SMART_ROUTER_ORCHESTRATOR=1`. They require only `smart-router` by default and both bundled skills when opted in, verify the saved session controls and native tool-result confirmation after each toggle, and explicitly request their children, including while routing is off. `test_integration_evidence.py` checks native tool-result extraction for both agents, including collapsed-output records, and excludes user echoes and assistant claims. -Dedicated regression coverage is missing for root-only orchestrator activation, +Dedicated regression coverage is missing for root-only Smart Router Orchestrator activation, compaction, retained skills in ineligible sessions, role-contract preservation, and isolation from legacy preference files. Codex's native hook merging and project trust, automatic delegation, and execution of @@ -194,7 +194,7 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_ug_codex_app_reports_unknown_argument` | Pass an invalid option directly to `ug codex app`, routing off/on | Real Codex parser error and status preserved | | `test_ug_codex_app_server_client_initializes` | Connect a stdio client, direct/`--` separator, routing off/on | Actual JSON-RPC initialize response; no non-JSON stdout; no routing | | `test_smart_routing_claude_route_subagent_hook`, `test_smart_routing_codex_route_subagent_hook` | Pipe a real PreToolUse spawn payload to the installed route-subagent hook with subagent-only routing enabled | Allow decision against the live router; requested model replaced by a routed agent definition (Claude) or bundled catalog slug (Codex) from the offered models; one audited decision matching the session and task | -| `test_smart_router_skill_toggles_claude_subagent_routing`, `test_smart_router_skill_toggles_codex_subagent_routing` | Configure, launch a real subagent-only TUI with orchestration unset or opted in, then spawn tagged children while invoking the installed Smart Router skill to switch routing on -> off -> on in the same session | Only `smart-router` is installed by default; opt-in also installs `orchestrate`; all three native children complete; only routing-enabled phases show the subagent banner and produce a live routing decision correlated with the child; no first-prompt routing wrapper; normal exit | +| `test_smart_router_skill_toggles_claude_subagent_routing`, `test_smart_router_skill_toggles_codex_subagent_routing` | Configure, launch a real subagent-only TUI with orchestration unset or opted in, then spawn tagged children while invoking the installed Smart Router skill to switch routing on -> off -> on in the same session | Only `smart-router` is installed by default; opt-in also installs `smart-router-orchestrator`; all three native children complete; only routing-enabled phases show the subagent banner and produce a live routing decision correlated with the child; no first-prompt routing wrapper; normal exit | | `test_ug_configure_claude_repeat_and_revert`, `test_ug_configure_codex_repeat_and_revert` | Configure twice over user settings; complete a task; revert twice | Settings preserved; no bearer in ug state; generated config removed; status unconfigured | | `test_ug_configure_claude_cleans_stale_skills_mcp_on_workspace_switch` | Configure the first workspace, register its skills MCP, switch to a second real workspace, and use Claude | Old registration removed from Claude and the new workspace state; old workspace bucket preserved; repeat configure stays clean; real file task completes on the second workspace | | `test_ug_configure_claude_rejects_invalid_credentials`, `test_ug_configure_codex_rejects_invalid_credentials` | Configure with a rejected bearer against the real workspace | Authentication failure; no successful saved setup | diff --git a/tests/integration/README.md b/tests/integration/README.md index 73a2736cd..a9333c1d7 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -281,9 +281,9 @@ PATH conflicts for the Smart Router skill have subprocess/component coverage in `ug` first in PATH. The live journeys above do not inject a second installation or establish PowerShell command execution. -The toggle journeys run with `ENABLE_ORCHESTRATION` unset and with -`ENABLE_ORCHESTRATION=1`. They require only `smart-router` by default and both -`orchestrate` and `smart-router` when opted in. They verify the saved session +The toggle journeys run with `ENABLE_SMART_ROUTER_ORCHESTRATOR` unset and with +`ENABLE_SMART_ROUTER_ORCHESTRATOR=1`. They require only `smart-router` by default and both +`smart-router-orchestrator` and `smart-router` when opted in. They verify the saved session controls, a new CLI confirmation in the native tool-result records, and a new assistant answer after each skill invocation. Collapsed terminal output is allowed; the answer need not repeat the CLI's exact wording. diff --git a/tests/integration/test_ug_smart_routing_hooks.py b/tests/integration/test_ug_smart_routing_hooks.py index 1cd6ffdf5..47fb3dd9b 100644 --- a/tests/integration/test_ug_smart_routing_hooks.py +++ b/tests/integration/test_ug_smart_routing_hooks.py @@ -126,8 +126,8 @@ def _toggle_with_skill(tui, session, agent: str, enabled: bool) -> None: if path.is_dir() and path.name not in ignored_skills ) expected_skills = ( - ["orchestrate", "smart-router"] - if session.env.get("ENABLE_ORCHESTRATION") == "1" + ["smart-router", "smart-router-orchestrator"] + if session.env.get("ENABLE_SMART_ROUTER_ORCHESTRATOR") == "1" else ["smart-router"] ) assert installed_skills == expected_skills, installed_skills @@ -291,11 +291,11 @@ def test_smart_router_skill_toggles_claude_subagent_routing( live_session, workspace, tmp_path, orchestration_enabled ): """Scenario: launch Claude with subagent routing enabled and orchestration unset - or opted in through ENABLE_ORCHESTRATION=1, spawn a child, invoke the + or opted in through ENABLE_SMART_ROUTER_ORCHESTRATOR=1, spawn a child, invoke the installed Smart Router skill to turn routing off, spawn another child, turn routing back on through the skill, and spawn a third child in the same real TUI session. - Expected: only Smart Router is installed by default; opting in also installs orchestrate. + Expected: only Smart Router is installed by default; opting in also installs smart-router-orchestrator. Each invocation records the CLI confirmation in the native transcript and changes the saved routing controls, even with collapsed terminal output; all three uniquely tagged calculations complete in native child sessions; only the first and third show the @@ -307,7 +307,7 @@ def test_smart_router_skill_toggles_claude_subagent_routing( session.env["ENABLE_SMART_ROUTING_V2"] = "1" session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" if orchestration_enabled: - session.env["ENABLE_ORCHESTRATION"] = "1" + session.env["ENABLE_SMART_ROUTER_ORCHESTRATOR"] = "1" config = build_coding_agent_config( "CODING_AGENT_CLAUDE_CODE", build_claude_agent_config(CLAUDE_SMART_ROUTING_MODELS, smart_routing=True), @@ -348,11 +348,11 @@ def test_smart_router_skill_toggles_codex_subagent_routing( live_session, workspace, tmp_path, orchestration_enabled ): """Scenario: launch Codex with subagent routing enabled and orchestration unset - or opted in through ENABLE_ORCHESTRATION=1, spawn a child, invoke the + or opted in through ENABLE_SMART_ROUTER_ORCHESTRATOR=1, spawn a child, invoke the installed Smart Router skill to turn routing off, spawn another child, turn routing back on through the skill, and spawn a third child in the same real TUI session. - Expected: only Smart Router is installed by default; opting in also installs orchestrate. + Expected: only Smart Router is installed by default; opting in also installs smart-router-orchestrator. Each invocation records the CLI confirmation in the native transcript and changes the saved routing controls, even with collapsed terminal output; all three uniquely tagged calculations complete in native child sessions; only the first and third show the @@ -364,7 +364,7 @@ def test_smart_router_skill_toggles_codex_subagent_routing( session.env["ENABLE_SMART_ROUTING_V2"] = "1" session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" if orchestration_enabled: - session.env["ENABLE_ORCHESTRATION"] = "1" + session.env["ENABLE_SMART_ROUTER_ORCHESTRATOR"] = "1" config = build_coding_agent_config( "CODING_AGENT_CODEX", build_codex_agent_config(models=CODEX_SMART_ROUTING_MODELS, smart_routing=True), From dd40db0240fbf4b518c55c70dcbf41462e666a29 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Thu, 8 Oct 2026 00:07:59 +0000 Subject: [PATCH 23/25] Show orchestration mode beside Claude subagents --- README.md | 4 ++++ src/ucode/smart_routing/v2.py | 4 +++- tests/test_claude_smart_routing_v2.py | 3 ++- 3 files changed, 9 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 2a93de5ec..72fa41694 100644 --- a/README.md +++ b/README.md @@ -262,6 +262,10 @@ to explorer, researcher, worker, tester, and reviewer roles while the root plans integrates, and verifies results. Easy tasks and explicit requests not to delegate stay in the root. +Claude's routing panel adds `[orchestrator on]` or `[orchestrator off]` to its +`Subagent` line. This reports the active session mode; it does not identify whether +a particular delegation came from the workflow or an explicit user request. + Once opted in, orchestration follows the existing smart-routing launch eligibility and session controls. Turning Smart Router off through its skill stops new automatic delegation; turning it on restores orchestration only in opted-in sessions. Explicit user diff --git a/src/ucode/smart_routing/v2.py b/src/ucode/smart_routing/v2.py index 0cd9205cf..ca2a0743f 100644 --- a/src/ucode/smart_routing/v2.py +++ b/src/ucode/smart_routing/v2.py @@ -412,10 +412,12 @@ def route_claude_pre_tool_use( route.decision, route.routed_model, ) + agent_name = claude_routing.SUBAGENT_NOTICE_CONFIG.name(route.tool_input) or "subagent" + orchestration = "on" if orchestrator.enabled() else "off" routing_message = claude_routing.SUBAGENT_NOTICE_CONFIG.message( route.decision, route.routed_model, - route.tool_input, + {**route.tool_input, "subagent_type": f"{agent_name} [orchestrator {orchestration}]"}, ) updated_input = { **{key: value for key, value in route.tool_input.items() if key != "model"}, diff --git a/tests/test_claude_smart_routing_v2.py b/tests/test_claude_smart_routing_v2.py index 6179e3a75..4e7b6fc0a 100644 --- a/tests/test_claude_smart_routing_v2.py +++ b/tests/test_claude_smart_routing_v2.py @@ -610,6 +610,7 @@ def fake_select(workspace, token, task, route_options, resolve, **kwargs): } def test_routes_agent_prompt_with_initialized_model_menu(self, tmp_path, monkeypatch): + monkeypatch.setenv("ENABLE_SMART_ROUTER_ORCHESTRATOR", "0") captured = {} decisions_path = tmp_path / "decisions.jsonl" monkeypatch.setattr(v2.claude_routing, "DECISIONS_PATH", decisions_path) @@ -665,7 +666,7 @@ def fake_select(workspace, token, task, route_options, resolve, **kwargs): expected_message = ( "\n┌───────────────────────────────────────────────────────────────────────────┐\n" "│ Using Unity Gateway Smart Router - Subagent │\n" - "│ Subagent : Explore │\n" + "│ Subagent : Explore [orchestrator off] │\n" "│ Selected Model : system.ai.claude-opus-4-8 │\n" "└───────────────────────────────────────────────────────────────────────────┘" ) From d2540a9e2780ae3889a2559ed7b8f2e3d05c4bd7 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Thu, 8 Oct 2026 00:22:11 +0000 Subject: [PATCH 24/25] Show orchestration mode beside Codex subagents --- README.md | 2 +- src/ucode/smart_routing/codex_routing.py | 8 ++++++-- src/ucode/smart_routing/routing.py | 6 +++++- tests/test_codex_routing.py | 15 ++++++++++++--- 4 files changed, 24 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index 72fa41694..01e0b5664 100644 --- a/README.md +++ b/README.md @@ -262,7 +262,7 @@ to explorer, researcher, worker, tester, and reviewer roles while the root plans integrates, and verifies results. Easy tasks and explicit requests not to delegate stay in the root. -Claude's routing panel adds `[orchestrator on]` or `[orchestrator off]` to its +Claude and Codex routing panels add `[orchestrator on]` or `[orchestrator off]` to the `Subagent` line. This reports the active session mode; it does not identify whether a particular delegation came from the workflow or an explicit user request. diff --git a/src/ucode/smart_routing/codex_routing.py b/src/ucode/smart_routing/codex_routing.py index 9569bc14d..0e3f09f98 100644 --- a/src/ucode/smart_routing/codex_routing.py +++ b/src/ucode/smart_routing/codex_routing.py @@ -15,10 +15,11 @@ # Python modules are singletons so patching this name patches the one call site. import urllib.request # noqa: F401 from collections.abc import Callable +from dataclasses import replace from typing import Any from ucode.config_io import APP_DIR -from ucode.smart_routing import routing +from ucode.smart_routing import orchestrator, routing from ucode.smart_routing.routing import RoutingDecision ROUTER_NAME = routing.ROUTER_NAME @@ -109,6 +110,7 @@ def route_pre_tool_use( def record(payload, task, decision, requested): routing.write_decision_record(DECISIONS_PATH, payload, task, decision, requested) + orchestration = "on" if orchestrator.enabled() else "off" return routing.route_spawn_tool( payload, is_spawn_agent=is_spawn_agent_tool, @@ -117,7 +119,9 @@ def record(payload, task, decision, requested): ), default_task_label="Codex subagent task", model_id_mapper=codex_model_id, - notice_config=SUBAGENT_NOTICE_CONFIG, + notice_config=replace( + SUBAGENT_NOTICE_CONFIG, name_suffix=f" [orchestrator {orchestration}]" + ), record_decision=record, ) diff --git a/src/ucode/smart_routing/routing.py b/src/ucode/smart_routing/routing.py index 3d11f3d33..cd6d7b499 100644 --- a/src/ucode/smart_routing/routing.py +++ b/src/ucode/smart_routing/routing.py @@ -134,6 +134,7 @@ class SubagentNoticeConfig: prompt_field: str display_model_mapper: Callable[[str], str] | None = None leading_newline: bool = False + name_suffix: str = "" def name(self, tool_input: dict[str, Any]) -> str | None: return _nonempty_string(tool_input.get(self.name_field)) @@ -153,10 +154,13 @@ def message( routed_model: str, tool_input: dict[str, Any], ) -> str: + subagent_name = self.name(tool_input) + if self.name_suffix: + subagent_name = f"{subagent_name or 'subagent'}{self.name_suffix}" message = decision.display_message( model_label=self.display_model(decision.model, routed_model), is_subagent=True, - subagent_name=self.name(tool_input), + subagent_name=subagent_name, prompt=self.prompt(tool_input), ) return f"\n{message}" if self.leading_newline else message diff --git a/tests/test_codex_routing.py b/tests/test_codex_routing.py index f18ea859f..0cd5f5740 100644 --- a/tests/test_codex_routing.py +++ b/tests/test_codex_routing.py @@ -153,6 +153,7 @@ def test_router_failure_fails_open(monkeypatch): def test_spawn_rewrite_preserves_original_input(monkeypatch): + monkeypatch.setenv("ENABLE_SMART_ROUTER_ORCHESTRATOR", "0") encrypted_message = {"encrypted": "opaque-ciphertext"} payload = { "tool_name": "collaborationspawn_agent", @@ -186,7 +187,7 @@ def test_spawn_rewrite_preserves_original_input(monkeypatch): expected_message = ( "\n┌───────────────────────────────────────────────────────────────────────────┐\n" "│ Using Unity Gateway Smart Router - Subagent │\n" - "│ Subagent : reviewer │\n" + "│ Subagent : reviewer [orchestrator off] │\n" "│ Selected Model : gpt-5.5 │\n" "│ Reason : Review needs deeper reasoning. │\n" "└───────────────────────────────────────────────────────────────────────────┘" @@ -204,7 +205,14 @@ def test_spawn_rewrite_preserves_original_input(monkeypatch): assert hook["permissionDecisionReason"] == expected_message -def test_spawn_rewrite_uses_codex_model_id_for_uc_endpoint(monkeypatch): +def test_spawn_rewrite_uses_codex_model_id_for_uc_endpoint(monkeypatch, tmp_path): + session_file = tmp_path / "env.json" + session_file.write_text("{}") + monkeypatch.setenv("UCODE_SESSION_ENV_FILE", str(session_file)) + monkeypatch.setenv("ENABLE_SMART_ROUTER_ORCHESTRATOR", "1") + monkeypatch.setenv("ENABLE_SMART_ROUTING_SUBAGENT_ONLY", "1") + monkeypatch.setenv("ENABLE_SMART_ROUTING_V2", "1") + monkeypatch.delenv("ISAAC_LAUNCH_MODE", raising=False) monkeypatch.setattr( codex_routing, "request_routing_decision", @@ -230,12 +238,13 @@ def test_spawn_rewrite_uses_codex_model_id_for_uc_endpoint(monkeypatch): expected_message = ( "\n┌───────────────────────────────────────────────────────────────────────────┐\n" "│ Using Unity Gateway Smart Router - Subagent │\n" - "│ Subagent : routing-smoke-test │\n" + "│ Subagent : routing-smoke-test [orchestrator on] │\n" "│ Selected Model : gpt-5.6-luna │\n" "└───────────────────────────────────────────────────────────────────────────┘" ) assert output["systemMessage"] == expected_message assert output["hookSpecificOutput"]["updatedInput"]["model"] == "gpt-5.6-luna" + assert output["hookSpecificOutput"]["updatedInput"]["task_name"] == "routing-smoke-test" def test_codex_model_id_maps_uc_gpt_models_to_codex_slugs(): From 56dbb51a271023dbcdc51760ecda95ffbf5e9ec0 Mon Sep 17 00:00:00 2001 From: Josh Joseph Date: Thu, 8 Oct 2026 00:25:50 +0000 Subject: [PATCH 25/25] Hide the orchestration label when disabled --- README.md | 7 ++++--- src/ucode/smart_routing/codex_routing.py | 6 ++---- src/ucode/smart_routing/v2.py | 5 +++-- tests/test_claude_smart_routing_v2.py | 2 +- tests/test_codex_routing.py | 2 +- 5 files changed, 11 insertions(+), 11 deletions(-) diff --git a/README.md b/README.md index 01e0b5664..d76cc92b4 100644 --- a/README.md +++ b/README.md @@ -262,9 +262,10 @@ to explorer, researcher, worker, tester, and reviewer roles while the root plans integrates, and verifies results. Easy tasks and explicit requests not to delegate stay in the root. -Claude and Codex routing panels add `[orchestrator on]` or `[orchestrator off]` to the -`Subagent` line. This reports the active session mode; it does not identify whether -a particular delegation came from the workflow or an explicit user request. +Claude and Codex routing panels add `[orchestrator on]` to the `Subagent` line +only when orchestration is active. When it is off, they show the normal subagent +name. The label reports the session mode; it does not identify whether a particular +delegation came from the workflow or an explicit user request. Once opted in, orchestration follows the existing smart-routing launch eligibility and session controls. Turning Smart Router off through its skill stops new automatic delegation; diff --git a/src/ucode/smart_routing/codex_routing.py b/src/ucode/smart_routing/codex_routing.py index 0e3f09f98..6d6137d28 100644 --- a/src/ucode/smart_routing/codex_routing.py +++ b/src/ucode/smart_routing/codex_routing.py @@ -110,7 +110,7 @@ def route_pre_tool_use( def record(payload, task, decision, requested): routing.write_decision_record(DECISIONS_PATH, payload, task, decision, requested) - orchestration = "on" if orchestrator.enabled() else "off" + name_suffix = " [orchestrator on]" if orchestrator.enabled() else "" return routing.route_spawn_tool( payload, is_spawn_agent=is_spawn_agent_tool, @@ -119,9 +119,7 @@ def record(payload, task, decision, requested): ), default_task_label="Codex subagent task", model_id_mapper=codex_model_id, - notice_config=replace( - SUBAGENT_NOTICE_CONFIG, name_suffix=f" [orchestrator {orchestration}]" - ), + notice_config=replace(SUBAGENT_NOTICE_CONFIG, name_suffix=name_suffix), record_decision=record, ) diff --git a/src/ucode/smart_routing/v2.py b/src/ucode/smart_routing/v2.py index ca2a0743f..5e4b1879b 100644 --- a/src/ucode/smart_routing/v2.py +++ b/src/ucode/smart_routing/v2.py @@ -413,11 +413,12 @@ def route_claude_pre_tool_use( route.routed_model, ) agent_name = claude_routing.SUBAGENT_NOTICE_CONFIG.name(route.tool_input) or "subagent" - orchestration = "on" if orchestrator.enabled() else "off" + if orchestrator.enabled(): + agent_name += " [orchestrator on]" routing_message = claude_routing.SUBAGENT_NOTICE_CONFIG.message( route.decision, route.routed_model, - {**route.tool_input, "subagent_type": f"{agent_name} [orchestrator {orchestration}]"}, + {**route.tool_input, "subagent_type": agent_name}, ) updated_input = { **{key: value for key, value in route.tool_input.items() if key != "model"}, diff --git a/tests/test_claude_smart_routing_v2.py b/tests/test_claude_smart_routing_v2.py index 4e7b6fc0a..103062622 100644 --- a/tests/test_claude_smart_routing_v2.py +++ b/tests/test_claude_smart_routing_v2.py @@ -666,7 +666,7 @@ def fake_select(workspace, token, task, route_options, resolve, **kwargs): expected_message = ( "\n┌───────────────────────────────────────────────────────────────────────────┐\n" "│ Using Unity Gateway Smart Router - Subagent │\n" - "│ Subagent : Explore [orchestrator off] │\n" + "│ Subagent : Explore │\n" "│ Selected Model : system.ai.claude-opus-4-8 │\n" "└───────────────────────────────────────────────────────────────────────────┘" ) diff --git a/tests/test_codex_routing.py b/tests/test_codex_routing.py index 0cd5f5740..9df7997ce 100644 --- a/tests/test_codex_routing.py +++ b/tests/test_codex_routing.py @@ -187,7 +187,7 @@ def test_spawn_rewrite_preserves_original_input(monkeypatch): expected_message = ( "\n┌───────────────────────────────────────────────────────────────────────────┐\n" "│ Using Unity Gateway Smart Router - Subagent │\n" - "│ Subagent : reviewer [orchestrator off] │\n" + "│ Subagent : reviewer │\n" "│ Selected Model : gpt-5.5 │\n" "│ Reason : Review needs deeper reasoning. │\n" "└───────────────────────────────────────────────────────────────────────────┘"