diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index 9738a8a4f..7124d45d1 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -392,8 +392,6 @@ jobs: UCODE_TEST_WORKSPACE: ${{ secrets.E2E_ADMIN_WORKSPACE }} DATABRICKS_CLIENT_ID: ${{ secrets.E2E_ADMIN_SP_CLIENT_ID }} DATABRICKS_CLIENT_SECRET: ${{ secrets.E2E_ADMIN_SP_CLIENT_SECRET }} - UG_MPS_DEFAULTS_CLIENT_SECRET: ${{ matrix.agent == 'claude' && secrets.UG_MPS_DEFAULTS_CLIENT_SECRET || '' }} - UG_PARENT_SCHEMA_DEFAULTS_CLIENT_SECRET: ${{ matrix.agent == 'claude' && secrets.UG_PARENT_SCHEMA_DEFAULTS_CLIENT_SECRET || '' }} UCODE_TEST_SECOND_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} DATABRICKS_SECOND_BEARER: ${{ secrets.DATABRICKS_BEARER }} run: | @@ -406,7 +404,7 @@ jobs: uv run --no-project --python 3.12 python scripts/run_integration.py \ --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ --default-index "$PACKAGE_INDEX" --output "$RUNNER_TEMP/ug-integration" \ - "${args[@]}" -- -m "(managed or managed_fixture or workspace_switch) and $AGENT" + "${args[@]}" -- -m "(managed_fixture or workspace_switch) and $AGENT" - name: Upload managed test evidence if: ${{ !cancelled() }} uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 diff --git a/scripts/run_integration.py b/scripts/run_integration.py index bc81650d3..8ea1532fd 100644 --- a/scripts/run_integration.py +++ b/scripts/run_integration.py @@ -31,20 +31,6 @@ "codex": "@openai/codex", "opencode": "opencode-ai", } -MANAGED_DEFAULTS_TARGETS = ( - ( - "UG_MPS_DEFAULTS_BEARER", - "https://eng-ml-inference-batch-inference-us-west-2.cloud.databricks.com", - "1c359c0f-58bc-42ac-a74f-079ccb173676", - "UG_MPS_DEFAULTS_CLIENT_SECRET", - ), - ( - "UG_PARENT_SCHEMA_DEFAULTS_BEARER", - "https://eng-ml-inference-ap-northeast-2.cloud.databricks.com", - "95e267dc-4393-4360-9d45-4b9b13b2d370", - "UG_PARENT_SCHEMA_DEFAULTS_CLIENT_SECRET", - ), -) WINDOWS_PATHEXT = ".COM;.EXE;.BAT;.CMD" UV_INDEX_CREDENTIAL_ENV = ( "UV_INDEX_DATABRICKS_PYPI_USERNAME", @@ -520,12 +506,7 @@ def terminate(signum, frame): bearer = os.environ.get("DATABRICKS_BEARER", "").strip() second_bearer = os.environ.get("DATABRICKS_SECOND_BEARER", "").strip() oauth_token = os.environ.get("CLAUDE_CODE_OAUTH_TOKEN", "").strip() - target_bearers: dict[str, str] = {} - client_secrets = ( - os.environ.get("DATABRICKS_CLIENT_SECRET", ""), - os.environ.get("UG_MPS_DEFAULTS_CLIENT_SECRET", ""), - os.environ.get("UG_PARENT_SCHEMA_DEFAULTS_CLIENT_SECRET", ""), - ) + client_secrets = (os.environ.get("DATABRICKS_CLIENT_SECRET", ""),) def redact(value: str) -> str: return redact_secrets( @@ -534,7 +515,6 @@ def redact(value: str) -> str: bearer, second_bearer, oauth_token, - *target_bearers.values(), *client_secrets, *installer_secrets, ), @@ -831,14 +811,6 @@ def run(command, *, cwd=output, env=base_env, timeout=600) -> str: if client_id and client_secret: bearer = mint_m2m_token(args.workspace, client_id, client_secret) - if not args.installation_only: - for bearer_env, target_workspace, client_id, secret_env in MANAGED_DEFAULTS_TARGETS: - secret = os.environ.get(secret_env, "").strip() - if args.workspace.rstrip("/") == target_workspace: - target_bearers[bearer_env] = bearer - elif secret: - target_bearers[bearer_env] = mint_m2m_token(target_workspace, client_id, secret) - test_dependencies = ["pytest==9.0.3"] if os.name == "posix": test_dependencies.extend(["pexpect==4.9.0", "pyte==0.8.2"]) @@ -876,10 +848,6 @@ def run(command, *, cwd=output, env=base_env, timeout=600) -> str: "UG_INTEGRATION_CODEX_PARENT_MODEL": args.codex_parent_model, "UCODE_TEST_WORKSPACE": args.workspace or "", "DATABRICKS_BEARER": bearer, - "UG_MPS_DEFAULTS_BEARER": target_bearers.get("UG_MPS_DEFAULTS_BEARER", ""), - "UG_PARENT_SCHEMA_DEFAULTS_BEARER": target_bearers.get( - "UG_PARENT_SCHEMA_DEFAULTS_BEARER", "" - ), "UCODE_TEST_SECOND_WORKSPACE": args.second_workspace or "", "DATABRICKS_SECOND_BEARER": second_bearer, "UG_INTEGRATION_WAREHOUSE_ID": args.warehouse_id or "", diff --git a/tests/AGENTS.md b/tests/AGENTS.md index 957e1c896..9a9c71fa9 100644 --- a/tests/AGENTS.md +++ b/tests/AGENTS.md @@ -29,11 +29,12 @@ tests. Keep work scoped to the behavior requested by the user. and regression coverage. Report failures instead of concealing them. The sole exception is the `managed_fixture` marker: it uses the built-in `UCODE_MANAGED_CONFIG_STUB` hook to inject the admin CodingAgentConfig INPUT so the - real `ug configure` path can be exercised across config shapes the live workspace does - not publish and one workspace fetch can be replayed across isolated cases. The gateway, - agent binaries, ug internals, and ug state stay real; the config fetch/wire contract stays - covered by the un-stubbed `managed` tests; and the hook must never be used to disable - validation or conceal a failure. + real `ug configure` path runs against a known config on the common managed workspace. + Every injected config is a checked-in JSON file in `fixtures/managed_config/` (the GET + shape of one CodingAgentConfig), so `ug configure --file` (#866) can replace the hook + mechanically. The gateway, agent binaries, ug internals, and ug state stay real; the + config fetch/wire contract is covered by unit tests and the read-only `e2e_cuj/` CUJs; + and the hook must never be used to disable validation or conceal a failure. 5. **Real responses and binaries.** Pin requested ug and agent versions. Never substitute a missing binary/service. Reuse explicit e2e workspace/auth settings; never pick a developer's Databricks profile automatically. diff --git a/tests/README.md b/tests/README.md index 66d24fa4f..5459ca961 100644 --- a/tests/README.md +++ b/tests/README.md @@ -184,7 +184,7 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_ug_configure_claude_repeat_and_revert`, `test_ug_configure_codex_repeat_and_revert` | Configure twice over user settings; complete a task; revert twice | Settings preserved; no bearer in ug state; generated config removed; status unconfigured | | `test_ug_configure_claude_cleans_stale_skills_mcp_on_workspace_switch` | Configure the first workspace, register its skills MCP, switch to a second real workspace, and use Claude | Old registration removed from Claude and the new workspace state; old workspace bucket preserved; repeat configure stays clean; real file task completes on the second workspace | | `test_ug_configure_claude_rejects_invalid_credentials`, `test_ug_configure_codex_rejects_invalid_credentials` | Configure with a rejected bearer against the real workspace | Authentication failure; no successful saved setup | -| `test_ug_configure_managed_claude`, `test_ug_configure_managed_codex` | Configure against a workspace that publishes a managed CodingAgentConfig | No agent selector; each agent's generated config exposes exactly the admin's static model_services; real gateway prompt on launch. The Codex case also checks the shared catalog pointer, restart guidance, and a fresh bare app-server's visible model list | +| `test_ug_configure_managed_claude`, `test_ug_configure_managed_codex` | Configure on the managed workspace with the stubbed `managed_workspace_default.json` CodingAgentConfig | No agent selector; each agent's generated config exposes exactly the admin's static model_services; real gateway prompt on launch. The Codex case also checks the shared catalog pointer, restart guidance, and a fresh bare app-server's visible model list | | `e2e_cuj/test_ug_budget_defaults.py` | Launch bare `ug` with separate low-spend and above-tier principals against the fixed 1% tier (`spending_percentage=0.01`); explicitly launch `ug claude` above the tier | Bare launches check Claude/Sonnet below the tier and the Codex/Luna recommendation above it; `ug usage` agrees with backend spend, threshold, and percentage. Explicit `ug claude` displays the backend's Codex/Luna recommendation while its generated model setting and native header select Sonnet. No budget writes or inference tasks. Run in the shared `E2E CUJs` job with both credential pairs documented in `integration/README.md`. | | `test_case_01_*` | Launch managed Claude without defaults after configure and from fresh state | Claude receives the admin MPS header; its gateway cache and replacement picker match the independently fetched provider model IDs; catalog labels are preserved and a model appears in a numbered picker row | | `test_case_03_*`, `test_case_05_*` | Pass a provider or model-location override to managed Claude after configure and from fresh state | ug rejects the override before Claude starts and preserves agent-owned state | @@ -192,7 +192,7 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_case_04_*`, `test_case_06_*` | Pass a provider or model-location override to managed Codex after configure and from fresh state | ug rejects the override before Codex starts and preserves agent-owned state | | `test_ug_configure_managed_codex_catalog_fallback` | Configure from an injected managed response containing a GPT model absent from Codex's bundled catalog | Actionable metadata warning; conservative catalog entry for the unknown model; real Codex prompt on the valid default model | | `test_managed_fixture_codex_http_headers_in_managed_file` | Interactive PTY configure with injected managed `http_headers` for Codex | The specified header (`x-databricks-workspace`) lands in `model_providers.Databricks.http_headers` in `/etc/codex/managed_config.toml` with the exact admin value | -| `test_managed_claude_mps_defaults_accompany_discovery`, `test_managed_claude_parent_schema_defaults_accompany_discovery` | Configure from the published admin config and launch Claude with MPS on `eng-ml-inference-batch-inference-us-west-2` and Unity Catalog discovery on `eng-ml-inference-ap-northeast-2`, respectively | Both generated settings files retain every admin-authored default alongside the source header and every independently fetched catalog model with its label; MPS pickers keep family shortcut rows separate from catalog entries; only UC Opus/Sonnet family ids gain `[1m]` | +| `test_managed_claude_mps_defaults_accompany_discovery`, `test_managed_claude_parent_schema_defaults_accompany_discovery` | Configure from a stubbed config and launch Claude with MPS discovery (`main.default.ci_e2e_anthropic_mps`) and with `system.ai` Unity Catalog discovery, respectively, both on the managed workspace | Both generated settings files retain every admin-authored default alongside the source header and every independently fetched catalog model with its label; MPS pickers keep family shortcut rows separate from catalog entries; only UC Opus/Sonnet family ids gain `[1m]` | | `test_unmanaged_claude_preserves_preexisting_family_defaults` | Seed Claude's OS-managed family defaults, then configure against one real workspace verified to have no managed config | Every pre-existing Claude family default remains unchanged in the OS-managed settings file | | `test_managed_fixture_claude_model_lifecycle`, `test_managed_fixture_codex_model_lifecycle` | Configure across no config -> static A -> static B -> MPS -> no config (stub-injected, `null` for no-config; MPS via a real provider service) | Each agent's model files reconcile to each static config (removed models pruned); switching to an MPS and a workspace with no managed config clears ug's static picker/catalog so no stale list is enforced | | `test_ug_installed_wheel_exposes_help_and_version` | Invoke freshly installed console command | Package version matches; public help works | @@ -202,13 +202,12 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi | `test_ug_and_ucode_web_search_helpers_preserve_mcp_stdio` | Initialize and list tools through both web-search helper commands | Exactly the MCP JSON-RPC responses; no text/ANSI contamination; existing server/tool identities preserved; no model request | With Claude and Codex selected there are **62 live cases** (12 marked TUI cases), -**6 managed-workspace cases** (marker `managed`, run against workspaces that -publish a CodingAgentConfig), **1 two-workspace case** (marker `workspace_switch`), -**25 managed-fixture cases** (marker `managed_fixture`, with only -the CodingAgentConfig input injected), and **7 installation checks**. The 14 retained numbered scenarios +**1 two-workspace case** (marker `workspace_switch`), +**33 managed-fixture cases** (marker `managed_fixture`, with only +the CodingAgentConfig input injected from a JSON file in `fixtures/managed_config/`), and **7 installation checks**. The 14 retained numbered scenarios comprise **24 explicit journeys**: 12 managed configured/fresh executions and 12 unmanaged -executions. Thirteen additional managed-fixture cases cover focused model, MCP, skills, -and lifecycle shapes; two published-config cases cover Claude defaults. Parametrization varies +executions. The remaining managed-fixture cases cover focused model, MCP, skills, +cache-TTL, and lifecycle shapes, including two Claude defaults cases. Parametrization varies argument spelling or routing mode, never hides the agent/provider in the test name. Duplicate boot-only cases are incorporated into the Databricks configuration TUI journeys. Generated-file cleanup and strict app-server stdout assertions remain enforced. diff --git a/tests/fixtures/managed_config/claude_lifecycle_a.json b/tests/fixtures/managed_config/claude_lifecycle_a.json new file mode 100644 index 000000000..f1d49f012 --- /dev/null +++ b/tests/fixtures/managed_config/claude_lifecycle_a.json @@ -0,0 +1,21 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CLAUDE_CODE", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CLAUDE_CODE", + "config": { + "models": { + "model_services": [ + "system.ai.claude-opus-4-8", + "system.ai.claude-sonnet-4-6", + "system.ai.claude-haiku-4-5" + ] + }, + "default_models": { + "default_model": "system.ai.claude-opus-4-8" + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/claude_lifecycle_b.json b/tests/fixtures/managed_config/claude_lifecycle_b.json new file mode 100644 index 000000000..54ec8f3a7 --- /dev/null +++ b/tests/fixtures/managed_config/claude_lifecycle_b.json @@ -0,0 +1,20 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CLAUDE_CODE", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CLAUDE_CODE", + "config": { + "models": { + "model_services": [ + "system.ai.claude-sonnet-4-6", + "system.ai.claude-haiku-4-5" + ] + }, + "default_models": { + "default_model": "system.ai.claude-sonnet-4-6" + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/claude_mcp.json b/tests/fixtures/managed_config/claude_mcp.json new file mode 100644 index 000000000..daf4629d5 --- /dev/null +++ b/tests/fixtures/managed_config/claude_mcp.json @@ -0,0 +1,24 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CLAUDE_CODE", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CLAUDE_CODE", + "config": { + "models": { + "model_services": [ + "system.ai.claude-opus-4-8" + ] + }, + "default_models": { + "default_model": "system.ai.claude-opus-4-8" + } + } + } + ], + "mcp_servers": { + "names": [ + "system.ai.github" + ] + } +} diff --git a/tests/fixtures/managed_config/claude_model_picker.json b/tests/fixtures/managed_config/claude_model_picker.json new file mode 100644 index 000000000..2a18a6ae7 --- /dev/null +++ b/tests/fixtures/managed_config/claude_model_picker.json @@ -0,0 +1,20 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CLAUDE_CODE", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CLAUDE_CODE", + "config": { + "models": { + "model_services": [ + "system.ai.claude-opus-4-8", + "system.ai.claude-sonnet-5" + ] + }, + "default_models": { + "default_model": "system.ai.claude-opus-4-8" + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/claude_mps.json b/tests/fixtures/managed_config/claude_mps.json new file mode 100644 index 000000000..0e1ab05b1 --- /dev/null +++ b/tests/fixtures/managed_config/claude_mps.json @@ -0,0 +1,14 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CLAUDE_CODE", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CLAUDE_CODE", + "config": { + "models": { + "model_provider_service": "main.default.ci_e2e_anthropic_mps" + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/claude_mps_defaults.json b/tests/fixtures/managed_config/claude_mps_defaults.json new file mode 100644 index 000000000..84b5736bc --- /dev/null +++ b/tests/fixtures/managed_config/claude_mps_defaults.json @@ -0,0 +1,21 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CLAUDE_CODE", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CLAUDE_CODE", + "config": { + "models": { + "model_provider_service": "main.default.ci_e2e_anthropic_mps" + }, + "default_models": { + "default_model": "anthropic.claude-sonnet-5", + "default_fable_model": "anthropic.claude-fable-5-1", + "default_opus_model": "anthropic.claude-opus-5", + "default_sonnet_model": "anthropic.claude-sonnet-5", + "default_haiku_model": "anthropic.claude-haiku-4-5" + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/claude_no_skills.json b/tests/fixtures/managed_config/claude_no_skills.json new file mode 100644 index 000000000..887bfcb5e --- /dev/null +++ b/tests/fixtures/managed_config/claude_no_skills.json @@ -0,0 +1,19 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CLAUDE_CODE", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CLAUDE_CODE", + "config": { + "models": { + "model_services": [ + "system.ai.claude-opus-4-8" + ] + }, + "default_models": { + "default_model": "system.ai.claude-opus-4-8" + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/claude_parent_schema_defaults.json b/tests/fixtures/managed_config/claude_parent_schema_defaults.json new file mode 100644 index 000000000..da90aa652 --- /dev/null +++ b/tests/fixtures/managed_config/claude_parent_schema_defaults.json @@ -0,0 +1,21 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CLAUDE_CODE", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CLAUDE_CODE", + "config": { + "models": { + "unity_catalog_location": "system.ai" + }, + "default_models": { + "default_model": "system.ai.claude-sonnet-5", + "default_fable_model": "system.ai.claude-fable-5-1", + "default_opus_model": "system.ai.claude-opus-5", + "default_sonnet_model": "system.ai.claude-sonnet-5", + "default_haiku_model": "system.ai.claude-haiku-4-5" + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/claude_skills_by_location.json b/tests/fixtures/managed_config/claude_skills_by_location.json new file mode 100644 index 000000000..30944afeb --- /dev/null +++ b/tests/fixtures/managed_config/claude_skills_by_location.json @@ -0,0 +1,22 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CLAUDE_CODE", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CLAUDE_CODE", + "config": { + "models": { + "model_services": [ + "system.ai.claude-opus-4-8" + ] + }, + "default_models": { + "default_model": "system.ai.claude-opus-4-8" + } + } + } + ], + "skills": { + "unity_catalog_location": "main.default" + } +} diff --git a/tests/fixtures/managed_config/claude_skills_by_name.json b/tests/fixtures/managed_config/claude_skills_by_name.json new file mode 100644 index 000000000..777ba766b --- /dev/null +++ b/tests/fixtures/managed_config/claude_skills_by_name.json @@ -0,0 +1,24 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CLAUDE_CODE", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CLAUDE_CODE", + "config": { + "models": { + "model_services": [ + "system.ai.claude-opus-4-8" + ] + }, + "default_models": { + "default_model": "system.ai.claude-opus-4-8" + } + } + } + ], + "skills": { + "names": [ + "main.default.forkable-meals" + ] + } +} diff --git a/tests/fixtures/managed_config/claude_smart_routing.json b/tests/fixtures/managed_config/claude_smart_routing.json new file mode 100644 index 000000000..dbb0bfb36 --- /dev/null +++ b/tests/fixtures/managed_config/claude_smart_routing.json @@ -0,0 +1,24 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CLAUDE_CODE", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CLAUDE_CODE", + "config": { + "models": { + "model_services": [ + "system.ai.claude-sonnet-5", + "system.ai.claude-haiku-4-5", + "system.ai.claude-opus-4-8" + ] + }, + "default_models": { + "default_model": "system.ai.claude-sonnet-5" + }, + "smart_routing": { + "enabled": true + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/claude_tracing.json b/tests/fixtures/managed_config/claude_tracing.json new file mode 100644 index 000000000..eca227a34 --- /dev/null +++ b/tests/fixtures/managed_config/claude_tracing.json @@ -0,0 +1,22 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CLAUDE_CODE", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CLAUDE_CODE", + "config": { + "models": { + "model_services": [ + "system.ai.claude-haiku-4-5" + ] + }, + "default_models": { + "default_model": "system.ai.claude-haiku-4-5" + }, + "tracing": { + "enabled": true + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/codex_catalog_fallback.json b/tests/fixtures/managed_config/codex_catalog_fallback.json new file mode 100644 index 000000000..ff4d51a0a --- /dev/null +++ b/tests/fixtures/managed_config/codex_catalog_fallback.json @@ -0,0 +1,20 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CODEX", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CODEX", + "config": { + "models": { + "model_services": [ + "system.ai.gpt-5-6-sol", + "system.ai.gpt-99" + ] + }, + "default_models": { + "default_model": "system.ai.gpt-5-6-sol" + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/codex_http_headers.json b/tests/fixtures/managed_config/codex_http_headers.json new file mode 100644 index 000000000..f7cb7563e --- /dev/null +++ b/tests/fixtures/managed_config/codex_http_headers.json @@ -0,0 +1,22 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CODEX", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CODEX", + "config": { + "models": { + "model_services": [ + "system.ai.gpt-5-6-sol" + ] + }, + "default_models": { + "default_model": "system.ai.gpt-5-6-sol" + }, + "http_headers": { + "x-databricks-workspace": "eng-ml-inference" + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/codex_lifecycle_a.json b/tests/fixtures/managed_config/codex_lifecycle_a.json new file mode 100644 index 000000000..1e8537a97 --- /dev/null +++ b/tests/fixtures/managed_config/codex_lifecycle_a.json @@ -0,0 +1,20 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CODEX", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CODEX", + "config": { + "models": { + "model_services": [ + "system.ai.gpt-5-6-sol", + "system.ai.gpt-5-4-nano" + ] + }, + "default_models": { + "default_model": "system.ai.gpt-5-6-sol" + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/codex_lifecycle_b.json b/tests/fixtures/managed_config/codex_lifecycle_b.json new file mode 100644 index 000000000..c1e23f9ef --- /dev/null +++ b/tests/fixtures/managed_config/codex_lifecycle_b.json @@ -0,0 +1,19 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CODEX", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CODEX", + "config": { + "models": { + "model_services": [ + "system.ai.gpt-5-4-nano" + ] + }, + "default_models": { + "default_model": "system.ai.gpt-5-4-nano" + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/codex_mcp.json b/tests/fixtures/managed_config/codex_mcp.json new file mode 100644 index 000000000..6cbb687a5 --- /dev/null +++ b/tests/fixtures/managed_config/codex_mcp.json @@ -0,0 +1,24 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CODEX", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CODEX", + "config": { + "models": { + "model_services": [ + "system.ai.gpt-5-6-sol" + ] + }, + "default_models": { + "default_model": "system.ai.gpt-5-6-sol" + } + } + } + ], + "mcp_servers": { + "names": [ + "system.ai.github" + ] + } +} diff --git a/tests/fixtures/managed_config/codex_mps.json b/tests/fixtures/managed_config/codex_mps.json new file mode 100644 index 000000000..095bcd55c --- /dev/null +++ b/tests/fixtures/managed_config/codex_mps.json @@ -0,0 +1,14 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CODEX", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CODEX", + "config": { + "models": { + "model_provider_service": "main.default.ci_e2e_openai_mps" + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/codex_skills_by_location.json b/tests/fixtures/managed_config/codex_skills_by_location.json new file mode 100644 index 000000000..c8f2666b8 --- /dev/null +++ b/tests/fixtures/managed_config/codex_skills_by_location.json @@ -0,0 +1,22 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CODEX", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CODEX", + "config": { + "models": { + "model_services": [ + "system.ai.gpt-5-6-sol" + ] + }, + "default_models": { + "default_model": "system.ai.gpt-5-6-sol" + } + } + } + ], + "skills": { + "unity_catalog_location": "main.default" + } +} diff --git a/tests/fixtures/managed_config/codex_smart_routing.json b/tests/fixtures/managed_config/codex_smart_routing.json new file mode 100644 index 000000000..32d41ff1a --- /dev/null +++ b/tests/fixtures/managed_config/codex_smart_routing.json @@ -0,0 +1,24 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CODEX", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CODEX", + "config": { + "models": { + "model_services": [ + "system.ai.gpt-5-6-sol", + "system.ai.gpt-5-6-terra", + "system.ai.gpt-5-6-luna" + ] + }, + "default_models": { + "default_model": "system.ai.gpt-5-6-sol" + }, + "smart_routing": { + "enabled": true + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/codex_tracing.json b/tests/fixtures/managed_config/codex_tracing.json new file mode 100644 index 000000000..03faf2668 --- /dev/null +++ b/tests/fixtures/managed_config/codex_tracing.json @@ -0,0 +1,22 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CODEX", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CODEX", + "config": { + "models": { + "model_services": [ + "system.ai.gpt-5-4-nano" + ] + }, + "default_models": { + "default_model": "system.ai.gpt-5-4-nano" + }, + "tracing": { + "enabled": true + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/managed_workspace_default.json b/tests/fixtures/managed_config/managed_workspace_default.json new file mode 100644 index 000000000..f47bb2860 --- /dev/null +++ b/tests/fixtures/managed_config/managed_workspace_default.json @@ -0,0 +1,34 @@ +{ + "spec_version": 1, + "default_agent": "CODING_AGENT_CLAUDE_CODE", + "enabled_agents": [ + { + "agent": "CODING_AGENT_CLAUDE_CODE", + "config": { + "models": { + "model_services": [ + "system.ai.claude-opus-4-8", + "system.ai.claude-sonnet-4-6", + "system.ai.claude-haiku-4-5" + ] + }, + "default_models": { + "default_model": "system.ai.claude-opus-4-8" + } + } + }, + { + "agent": "CODING_AGENT_CODEX", + "config": { + "models": { + "model_services": [ + "system.ai.gpt-5-6-sol" + ] + }, + "default_models": { + "default_model": "system.ai.gpt-5-6-sol" + } + } + } + ] +} diff --git a/tests/fixtures/managed_config/no_config.json b/tests/fixtures/managed_config/no_config.json new file mode 100644 index 000000000..19765bd50 --- /dev/null +++ b/tests/fixtures/managed_config/no_config.json @@ -0,0 +1 @@ +null diff --git a/tests/integration/README.md b/tests/integration/README.md index 2cb2bcbbb..14ebef990 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -229,14 +229,14 @@ test_ug_smart_routing_hooks.py # live hook contract plus skill-driven test_ug_configure_claude_lifecycle.py # repeat setup, revert, rejected credentials test_ug_configure_claude_workspace_switch.py # real skills MCP cleanup across two workspaces test_ug_configure_codex_lifecycle.py # repeat setup, revert, rejected credentials -test_ug_claude_managed_model_discovery.py # fetched/reused Claude MPS policy cases -test_ug_codex_managed_model_discovery.py # fetched/reused Codex MPS policy cases +test_ug_claude_managed_model_discovery.py # Claude MPS policy cases (claude_mps.json) +test_ug_codex_managed_model_discovery.py # Codex MPS policy cases (codex_mps.json) test_ug_claude_model_discovery.py # unmanaged scenarios 7, 9, 11, 13 test_ug_codex_model_discovery.py # unmanaged scenarios 8, 10, 12, 14 -test_ug_configure_managed.py # managed workspace: static model list/catalog pointer, no agent selector -test_ug_configure_managed_models.py # injected model sources, smart-routing banner, Codex fallback metadata -test_ug_configure_managed_mcp.py # injected managed MCP list -test_ug_configure_managed_skills.py # injected managed skills: download, coexist, reconcile away +test_ug_configure_managed.py # stubbed static model list/catalog pointer, no agent selector, cache TTL +test_ug_configure_managed_models.py # stubbed model sources, smart-routing banner, Codex fallback metadata +test_ug_configure_managed_mcp.py # stubbed managed MCP list +test_ug_configure_managed_skills.py # stubbed managed skills: download, coexist, reconcile away test_ug_configure_managed_lifecycle.py # none -> A -> B -> MPS -> none: reconcile, clear on MPS/no-config test_installation.py # fresh installed package utils/ # process/terminal/evidence helpers and Docker files @@ -308,8 +308,8 @@ fails the selected CUJ, rather than skipping it. The tracing journeys are part of their respective Full agent lanes and use the existing e2e workspace and bearer. Because that workspace deliberately has no published managed -configuration, each journey injects only a tracing-enabled CodingAgentConfig input through -the suite's managed-config stub seam; the agents, inference, OTLP export, and table +configuration, each journey injects only a tracing-enabled CodingAgentConfig input (the +`claude_tracing.json` / `codex_tracing.json` fixtures) through the suite's managed-config stub; the agents, inference, OTLP export, and table verification remain real. Each adds the prompt's UUID as a trace-safe `ug_integration_marker` attribute, resolves the destination table from the workspace tracing configuration, waits 30 seconds, and queries that table through an existing SQL warehouse. @@ -363,30 +363,24 @@ they only configure, list models, and open/close the picker. Other live CUJs per real model tasks. There are **62 live cases** (including 12 marked TUI journeys) and **7 installation -checks** with Claude and Codex; selecting OpenCode adds one live headless case. A separate **6 managed-workspace cases** (one per agent, an idempotent -re-configure, a cache-TTL journey, and two Claude defaults cases; marker `managed`) run against -workspaces that publish CodingAgentConfigs; see "Managed-workspace journeys" below. One **`workspace_switch` case** -uses two real workspaces and checks skills MCP cleanup and a completed Claude task. -A further **25 `managed_fixture` -cases** use `UCODE_MANAGED_CONFIG_STUB`. Twelve explicit configured/fresh Claude and Codex -discovery and source-override journeys fetch the published config once per agent, replace that -agent's static source with its dedicated MPS, and reuse the result. Thirteen other collected cases -cover focused model, MCP, skills, and lifecycle shapes, including per-agent model reconciliation -and managed skill cleanup. The two Claude default-model cases read published MPS and Unity -Catalog sources directly from `eng-ml-inference-batch-inference-us-west-2` and -`eng-ml-inference-ap-northeast-2`, respectively, then verify both generated settings files retain -all admin-authored family defaults. Their replacement pickers contain those mapped defaults plus +checks** with Claude and Codex; selecting OpenCode adds one live headless case. One +**`workspace_switch` case** uses two real workspaces and checks skills MCP cleanup and a completed +Claude task. A further **33 `managed_fixture` cases** (two of them also `live`) run on the +managed workspace with a checked-in JSON CodingAgentConfig from `tests/fixtures/managed_config/` +injected through `UCODE_MANAGED_CONFIG_STUB`; there are no cases that read a published config. +Twelve explicit configured/fresh Claude and Codex discovery and source-override journeys use the +`claude_mps.json` / `codex_mps.json` fixtures, which set each agent's source to its dedicated MPS. +The other cases cover focused model, MCP, skills, cache-TTL, and lifecycle shapes, including +per-agent model reconciliation and managed skill cleanup. Of the two Claude default-model cases, +the MPS case stubs a `main.default.ci_e2e_anthropic_mps` source and the `system.ai` parent-schema +case stubs a Unity Catalog source; both verify the generated settings +files retain all admin-authored family defaults. Their replacement pickers contain those mapped defaults plus the independently fetched MPS or UC schema catalog. MPS family shortcut rows remain separate from catalog rows for the same target; UC model IDs are deduplicated and catalog labels are retained. Direct renderer tests cover default/catalog composition, while focused CLI regressions verify UC catalog discovery with overall defaults, family defaults, or both, along with explicit model -selection and preservation of static model lists. Neither case injects -a config. Each obtains a token for its -target workspace using OAuth client credentials. The two target service-principal client IDs are -constants in the runner; CI only needs `UG_MPS_DEFAULTS_CLIENT_SECRET` for west-2 and -`UG_PARENT_SCHEMA_DEFAULTS_CLIENT_SECRET` for northeast-2. As with the base workspace, the runner -mints short-lived tokens and passes bearers to pytest; each test selects its target bearer for -`ug configure` and Claude. The client secrets do not enter the pytest process. +selection and preservation of static model lists. Both cases run on the managed workspace's own +bearer; no second workspace or extra secret is involved. The 14 retained numbered scenarios comprise 24 explicit journeys: 12 managed and 12 unmanaged executions; the complete integration suite collects 101 executions. See the named coverage and gaps matrix in [../README.md](../README.md). @@ -564,21 +558,23 @@ cannot still be running when that gate passes. Full coverage on PRs needs no lab ### Managed-workspace journeys -`test_ug_configure_managed.py` (marker `managed`, not `live`) runs in its own per-agent -**Managed config** jobs against a second workspace that publishes an admin CodingAgentConfig, -whereas unmanaged live cases require a workspace without one. `ug configure` applies the admin config -with no agent selector, and each agent's generated config exposes exactly the admin's static -`model_services` (Claude's `availableModels`/`modelPicker`, Codex's model catalog). The managed -Codex case also checks stderr guidance to restart the daemon after publication, -that the shared app config points at the stable catalog, and that a fresh -bare Codex app-server returns the expected visible model before the existing TUI prompt/input -assertion. It does not claim GUI rendering or inference coverage. - -Two `managed` cases in `test_ug_configure_managed_models.py` target separate published configs: -west-2 must publish a Claude MPS source with Anthropic family defaults, and northeast-2 must -publish a Claude `system.ai` parent-schema source with Unity Catalog family defaults. `ug configure` -fetches the config; the tests assert the generated private and OS-managed Claude settings after -launch. Their exact required defaults and source headers are reflected in the tests. +The managed lanes run `managed_fixture` journeys (and `workspace_switch`) on the single managed +workspace (`E2E_ADMIN_WORKSPACE`), in per-agent **Managed config** jobs. Every managed config comes +from a JSON file in `tests/fixtures/managed_config/` in the GET shape of one CodingAgentConfig; a +fixture whose content is JSON `null` stands for a workspace with no managed config. Tests select a +fixture with `use_managed_config_fixture(session, name)`, which points `UCODE_MANAGED_CONFIG_STUB` +at the file. +`ug configure --file` (#866) will replace the environment stub; because the inputs are already +checked-in files, that switch is mechanical. + +`test_ug_configure_managed.py` uses `managed_workspace_default.json`. `ug configure` applies the +config with no agent selector, and each agent's generated config exposes exactly its static +`model_services` (Claude's `availableModels`/`modelPicker`, Codex's model catalog). The stub takes +the same cache path as a live read (`published` outcome, `retrieved_at` stamp, TTL), so the +cache-TTL journey holds unchanged. The Codex case also checks stderr guidance to restart the +daemon after publication, that the shared app config points at the stable catalog, and that a +fresh bare Codex app-server returns the expected visible model before the existing TUI +prompt/input assertion. It does not claim GUI rendering or inference coverage. `test_unmanaged_claude_preserves_preexisting_family_defaults` is a live lifecycle journey against one real workspace. A read-only check first proves that the workspace publishes no @@ -586,16 +582,15 @@ CodingAgentConfig. The test then seeds `/etc/claude-code/managed-settings.json`, `ug configure`, and requires every pre-existing family default to survive exactly. It checks settings reconciliation and makes no model-inference claim. -Treat that published CodingAgentConfig as shared CI fixture state. The managed lanes assert its -exact model ids and its both-agent enablement, so editing the managed workspace's config (models, -enabled agents, or defaults) breaks these lanes until the constants in `test_ug_configure_managed.py` -are updated to match. Do not change it casually. +The expected model ids in `test_ug_configure_managed.py` mirror `managed_workspace_default.json`; +update both together. The fixtures reference real securables on the managed workspace (MPS +`main.default.ci_e2e_anthropic_mps` / `ci_e2e_openai_mps`, MCP service `system.ai.github`, skill +`main.default.forkable-meals`), which must keep existing. The `managed_fixture` journeys use `UCODE_MANAGED_CONFIG_STUB` to short-circuit only the -managed-config HTTP read for config shapes that workspace does not publish. The Claude discovery -module fetches the workspace's published config once, replaces Claude's static model source with -`main.default.ci_e2e_anthropic_mps`, drops incompatible static defaults, and reuses that fixture -across all configured/fresh scenarios. The Codex module does the same with +managed-config HTTP read. The Claude discovery module uses `claude_mps.json`, whose Claude model +source is `main.default.ci_e2e_anthropic_mps` with no static defaults, across all configured/fresh +scenarios. The Codex module does the same with `codex_mps.json` and `main.default.ci_e2e_openai_mps`. Separate read-only, provider-scoped model-list requests establish expected IDs independently of the generated agent files. With no authored defaults, Claude's native cache and replacement picker must match those IDs, preserve catalog display names, @@ -626,23 +621,16 @@ these same-repository secrets rather than storing a long-lived bearer: - `E2E_ADMIN_SP_CLIENT_ID` / `E2E_ADMIN_SP_CLIENT_SECRET`: the service principal's OAuth client credentials. The job passes them to the runner as the standard `DATABRICKS_CLIENT_ID` / `DATABRICKS_CLIENT_SECRET`, and `run_integration.py` mints the workspace token. -- `UG_MPS_DEFAULTS_CLIENT_SECRET`: OAuth client secret for the west-2 Claude defaults workspace. - Its client ID (`1c359c0f-58bc-42ac-a74f-079ccb173676`) is in the runner code. -- `UG_PARENT_SCHEMA_DEFAULTS_CLIENT_SECRET`: OAuth client secret for the northeast-2 Claude - defaults workspace. Its client ID (`95e267dc-4393-4360-9d45-4b9b13b2d370`) is in the runner code. - CI passes these two repository secrets only to the managed Claude lane. Run it locally the same way, pointing at the managed workspace: ```bash export UCODE_TEST_WORKSPACE=https:// export DATABRICKS_CLIENT_ID= DATABRICKS_CLIENT_SECRET= -python scripts/run_integration.py --claude-version --codex-version -- -m managed +python scripts/run_integration.py --claude-version --codex-version -- -m managed_fixture ``` -To run only the two Claude defaults cases locally, set the base managed workspace and its -`DATABRICKS_CLIENT_ID` / `DATABRICKS_CLIENT_SECRET` as above, set both target-specific client -secrets, and select the tests by name: +To run only the two Claude defaults cases locally, select them by name with the same environment: ```bash python3.12 scripts/run_integration.py \ diff --git a/tests/integration/pytest.ini b/tests/integration/pytest.ini index cc0d9b0ca..b997f99be 100644 --- a/tests/integration/pytest.ini +++ b/tests/integration/pytest.ini @@ -4,8 +4,7 @@ addopts = --strict-markers --tb=short markers = installation: installed-package checks that require no workspace live: requires the real workspace used by the existing e2e suite - managed: requires an admin CodingAgentConfig published on the selected workspace - managed_fixture: real ug/TUI against a real workspace, but the managed CodingAgentConfig is injected via UCODE_MANAGED_CONFIG_STUB + managed_fixture: real ug/TUI against a real workspace, with the managed CodingAgentConfig injected from a tests/fixtures/managed_config JSON file via UCODE_MANAGED_CONFIG_STUB workspace_switch: requires two real workspace hosts and their respective credentials smoke: Databricks Hosted, custom OAuth CLI TUI, and headless prompt for each agent tui: real interactive terminal boot, keyboard input, exit and reopen diff --git a/tests/integration/test_ug_claude_managed_model_discovery.py b/tests/integration/test_ug_claude_managed_model_discovery.py index f674d225e..cc6ad956f 100644 --- a/tests/integration/test_ug_claude_managed_model_discovery.py +++ b/tests/integration/test_ug_claude_managed_model_discovery.py @@ -1,8 +1,8 @@ """Claude managed-config CUJs for repository scenarios 1, 3, and 5. -The admin CodingAgentConfig is fetched once from the managed workspace, its Claude model source is -set to the dedicated test MPS, and the result is reused through ``UCODE_MANAGED_CONFIG_STUB`` in -each isolated session. Normalization, config writers, the gateway, and Claude Code remain real. +The admin CodingAgentConfig is the checked-in ``claude_mps.json`` fixture, whose Claude model source +is the dedicated test MPS, injected through ``UCODE_MANAGED_CONFIG_STUB`` in each isolated session. +Normalization, config writers, the gateway, and Claude remain real. """ import json @@ -11,9 +11,8 @@ import pytest from utils.constants import MANAGED_CLAUDE_PROVIDER_SERVICE from utils.managed import ( - fetch_managed_config_stub, is_managed_config_control_plane_cache, - use_managed_config_stub, + use_managed_config_fixture, ) from utils.model_discovery import claude_model_in_picker from utils.provider_catalog import ( @@ -25,18 +24,6 @@ pytestmark = [pytest.mark.managed_fixture, pytest.mark.claude] -@pytest.fixture(scope="module") -def _managed_claude_config_stub(workspace, tmp_path_factory): - return fetch_managed_config_stub( - workspace, - os.environ["DATABRICKS_BEARER"], - tmp_path_factory.mktemp("managed-config-claude"), - "managed-config-claude.json", - agent="CODING_AGENT_CLAUDE_CODE", - provider_service=MANAGED_CLAUDE_PROVIDER_SERVICE, - ) - - @pytest.fixture(scope="module") def _managed_claude_provider_catalog(workspace): return fetch_anthropic_provider_catalog( @@ -47,8 +34,8 @@ def _managed_claude_provider_catalog(workspace): @pytest.fixture(autouse=True) -def _managed_claude_config(live_session, _managed_claude_config_stub): - use_managed_config_stub(live_session, _managed_claude_config_stub) +def _managed_claude_config(live_session): + use_managed_config_fixture(live_session, "claude_mps") def _claude_state_and_agent_files(session): diff --git a/tests/integration/test_ug_claude_tracing.py b/tests/integration/test_ug_claude_tracing.py index ecc8b8f9a..5618298b3 100644 --- a/tests/integration/test_ug_claude_tracing.py +++ b/tests/integration/test_ug_claude_tracing.py @@ -7,17 +7,13 @@ import pytest from utils.constants import CLAUDE_TEST_MODEL from utils.evidence import FileTask -from utils.managed import ( - build_claude_agent_config, - build_coding_agent_config, - set_managed_config_stub, -) +from utils.managed import use_managed_config_fixture from utils.sql import query_count, resolve_trace_table, resolve_warehouse_id pytestmark = [pytest.mark.live, pytest.mark.claude] -def test_ug_claude_exports_trace_to_configured_table(live_session, workspace, tmp_path): +def test_ug_claude_exports_trace_to_configured_table(live_session, workspace): """Scenario: resolve the trace table, configure Claude, and run a uniquely marked task. Expected: the real agent task completes and, after the ingestion window, the @@ -30,11 +26,7 @@ def test_ug_claude_exports_trace_to_configured_table(live_session, workspace, tm warehouse_id = warehouse_id or resolve_warehouse_id(workspace, bearer) marker = f"ug-claude-trace-{uuid.uuid4().hex}" task = FileTask(session) - config = build_coding_agent_config( - "CODING_AGENT_CLAUDE_CODE", - build_claude_agent_config([CLAUDE_TEST_MODEL], otel_tracing_enabled=True), - ) - set_managed_config_stub(session, tmp_path, config) + use_managed_config_fixture(session, "claude_tracing") session.run( "configure", "--workspace", diff --git a/tests/integration/test_ug_codex_managed_model_discovery.py b/tests/integration/test_ug_codex_managed_model_discovery.py index 980f69085..1aeadeab9 100644 --- a/tests/integration/test_ug_codex_managed_model_discovery.py +++ b/tests/integration/test_ug_codex_managed_model_discovery.py @@ -1,8 +1,8 @@ """Codex managed-config CUJs for repository scenarios 2, 4, and 6. -The admin CodingAgentConfig is fetched once from the managed workspace, its Codex model source is -set to the dedicated test MPS, and the result is reused through ``UCODE_MANAGED_CONFIG_STUB`` in -each isolated session. Normalization, config writers, the gateway, and Codex remain real. +The admin CodingAgentConfig is the checked-in ``codex_mps.json`` fixture, whose Codex model source +is the dedicated test MPS, injected through ``UCODE_MANAGED_CONFIG_STUB`` in each isolated session. +Normalization, config writers, the gateway, and Codex remain real. """ import json @@ -12,9 +12,8 @@ import pytest from utils.constants import MANAGED_CODEX_PROVIDER_SERVICE from utils.managed import ( - fetch_managed_config_stub, is_managed_config_control_plane_cache, - use_managed_config_stub, + use_managed_config_fixture, ) from utils.provider_catalog import ( CodexProviderCatalog, @@ -26,18 +25,6 @@ pytestmark = [pytest.mark.managed_fixture, pytest.mark.codex] -@pytest.fixture(scope="module") -def _managed_codex_config_stub(workspace, tmp_path_factory): - return fetch_managed_config_stub( - workspace, - os.environ["DATABRICKS_BEARER"], - tmp_path_factory.mktemp("managed-config-codex"), - "managed-config-codex.json", - agent="CODING_AGENT_CODEX", - provider_service=MANAGED_CODEX_PROVIDER_SERVICE, - ) - - @pytest.fixture(scope="module") def _managed_codex_provider_catalog(workspace): return fetch_codex_provider_catalog( @@ -48,8 +35,8 @@ def _managed_codex_provider_catalog(workspace): @pytest.fixture(autouse=True) -def _managed_codex_config(live_session, _managed_codex_config_stub): - use_managed_config_stub(live_session, _managed_codex_config_stub) +def _managed_codex_config(live_session): + use_managed_config_fixture(live_session, "codex_mps") def _codex_state_and_agent_files(session): diff --git a/tests/integration/test_ug_codex_tracing.py b/tests/integration/test_ug_codex_tracing.py index be935837e..bfaec6d8c 100644 --- a/tests/integration/test_ug_codex_tracing.py +++ b/tests/integration/test_ug_codex_tracing.py @@ -7,17 +7,13 @@ import pytest from utils.constants import CODEX_TEST_MODEL from utils.evidence import FileTask -from utils.managed import ( - build_codex_agent_config, - build_coding_agent_config, - set_managed_config_stub, -) +from utils.managed import use_managed_config_fixture from utils.sql import query_count, resolve_trace_table, resolve_warehouse_id pytestmark = [pytest.mark.live, pytest.mark.codex] -def test_ug_codex_exports_trace_to_configured_table(live_session, workspace, tmp_path): +def test_ug_codex_exports_trace_to_configured_table(live_session, workspace): """Scenario: resolve the trace table, configure Codex, and run a uniquely marked task. Expected: the real agent task completes and, after the ingestion window, the @@ -30,11 +26,7 @@ def test_ug_codex_exports_trace_to_configured_table(live_session, workspace, tmp warehouse_id = warehouse_id or resolve_warehouse_id(workspace, bearer) marker = f"ug-codex-trace-{uuid.uuid4().hex}" task = FileTask(session) - config = build_coding_agent_config( - "CODING_AGENT_CODEX", - build_codex_agent_config(models=[CODEX_TEST_MODEL], otel_tracing_enabled=True), - ) - set_managed_config_stub(session, tmp_path, config) + use_managed_config_fixture(session, "codex_tracing") session.run( "configure", "--workspace", diff --git a/tests/integration/test_ug_configure_managed.py b/tests/integration/test_ug_configure_managed.py index 85ac2540b..0d4709c99 100644 --- a/tests/integration/test_ug_configure_managed.py +++ b/tests/integration/test_ug_configure_managed.py @@ -1,17 +1,18 @@ """CUJs: configure against a managed workspace, where an admin publishes the setup. -These run against the managed e2e workspace (`E2E_ADMIN_WORKSPACE`), which publishes a -CodingAgentConfig. They exercise the managed config fetch end to end: -`ug configure` applies the admin config to every enabled agent without the personal agent -selector, and each agent's generated config exposes exactly the admin's static -`model_services` (Claude's `availableModels`/`modelPicker`, Codex's model catalog). The -expected model ids mirror the published config; update them here if the admin list changes. +These run against the managed e2e workspace (`E2E_ADMIN_WORKSPACE`) with the checked-in +`managed_workspace_default.json` CodingAgentConfig injected via UCODE_MANAGED_CONFIG_STUB, so they +do not depend on what the workspace happens to publish. `ug configure` applies the config to every +enabled agent without the personal agent selector, and each agent's generated config exposes +exactly the config's static `model_services` (Claude's `availableModels`/`modelPicker`, Codex's +model catalog). The expected model ids mirror that fixture; update both together. """ import json import tomllib import pytest +from utils.managed import use_managed_config_fixture from utils.terminal import AgentTerminal MANAGED_CLAUDE_MODELS = [ @@ -22,10 +23,10 @@ MANAGED_CODEX_MODEL = "system.ai.gpt-5-6-sol" -@pytest.mark.managed +@pytest.mark.managed_fixture @pytest.mark.claude def test_ug_configure_managed_claude(live_session, workspace): - """Scenario: run `ug configure` on a workspace that publishes a managed config. + """Scenario: run `ug configure` under an injected managed config. Expected: ug applies the admin config to every enabled agent without showing the personal agent selector, Claude's generated settings expose exactly the admin's static @@ -33,6 +34,7 @@ def test_ug_configure_managed_claude(live_session, workspace): prompt rather than the account-login flow. """ session = live_session + use_managed_config_fixture(session, "managed_workspace_default") result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) assert "Select coding agents to configure:" not in result.stdout, result.stdout @@ -48,10 +50,10 @@ def test_ug_configure_managed_claude(live_session, workspace): tui.check_input_and_exit() -@pytest.mark.managed +@pytest.mark.managed_fixture @pytest.mark.codex def test_ug_configure_managed_codex(live_session, workspace): - """Scenario: run `ug configure` on a workspace that publishes a managed config. + """Scenario: run `ug configure` under an injected managed config. Expected: ug applies the admin config to every enabled agent without showing the personal agent selector, Codex's generated model catalog lists exactly the admin's static @@ -61,6 +63,7 @@ def test_ug_configure_managed_codex(live_session, workspace): a prompt, accepts input, and exits normally. GUI rendering and inference are not covered. """ session = live_session + use_managed_config_fixture(session, "managed_workspace_default") result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) assert "Select coding agents to configure:" not in result.stdout, result.stdout assert "codex app-server daemon restart" in " ".join(result.stderr.split()), result.stderr @@ -89,7 +92,7 @@ def test_ug_configure_managed_codex(live_session, workspace): tui.check_input_and_exit() -@pytest.mark.managed +@pytest.mark.managed_fixture @pytest.mark.claude def test_ug_configure_managed_is_idempotent(live_session, workspace): """Scenario: run the managed `ug configure` twice in the same session. @@ -98,6 +101,7 @@ def test_ug_configure_managed_is_idempotent(live_session, workspace): agents identically, so a repeat configure neither duplicates, drops, nor rewrites any entry. """ session = live_session + use_managed_config_fixture(session, "managed_workspace_default") runs = [] for _ in range(2): result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) @@ -112,18 +116,19 @@ def test_ug_configure_managed_is_idempotent(live_session, workspace): assert runs == [expected, expected], runs -@pytest.mark.managed +@pytest.mark.managed_fixture @pytest.mark.claude def test_ug_managed_config_launch_reuses_cache_within_ttl(live_session, workspace): """Scenario: after a managed `ug configure`, launch Claude within the cache TTL, then again after the cached read is backdated past it. Expected: configure stamps managed-config.json with a `published` outcome and a `retrieved_at`; - a launch within the TTL is served from that cache and leaves the stamp untouched (no - control-plane re-read), and once the stamp is backdated past the TTL the next launch reads fresh - and advances it. + a launch within the TTL is served from that cache and leaves the stamp untouched (no re-read of + the config source), and once the stamp is backdated past the TTL the next launch reads fresh + and advances it. The stub takes the same cache path as a live read. """ session = live_session + use_managed_config_fixture(session, "managed_workspace_default") cache = session.home / ".ucode" / "managed-config.json" session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) diff --git a/tests/integration/test_ug_configure_managed_lifecycle.py b/tests/integration/test_ug_configure_managed_lifecycle.py index b2fba910b..d14325d5c 100644 --- a/tests/integration/test_ug_configure_managed_lifecycle.py +++ b/tests/integration/test_ug_configure_managed_lifecycle.py @@ -7,23 +7,18 @@ configuring a workspace with no managed config clears ug's static picker/catalog so an unmanaged workspace never enforces a stale list. -The configs are injected via ``UCODE_MANAGED_CONFIG_STUB`` (an explicit ``null`` for the no-config -states) so the transitions run without republishing the live workspace's config; auth, the config -writers, the agent binaries, and the workspace stay real. The MPS states reference real provider -services published on the managed e2e workspace. These assert the generated files because the whole -point is file-level reconciliation across transitions, which the TUI cannot show. +The configs are checked-in JSON fixtures (tests/fixtures/managed_config/; ``no_config`` is an +explicit ``null``) injected via ``UCODE_MANAGED_CONFIG_STUB`` so the transitions run without +republishing the managed workspace's config; auth, the config writers, the agent binaries, and the +workspace stay real. The MPS fixtures reference real provider services on that workspace. These +assert the generated files because the whole point is file-level reconciliation across +transitions, which the TUI cannot show. """ import json import pytest -from utils.managed import ( - build_claude_agent_config, - build_codex_agent_config, - build_coding_agent_config, - build_mps_agent_config, - set_managed_config_stub, -) +from utils.managed import use_managed_config_fixture # Claude model ids are written to the picker verbatim, so any real system.ai ids work. CLAUDE_A = [ @@ -35,9 +30,6 @@ # Codex ids must survive the bundled-catalog build; both of these are used elsewhere in the suite. CODEX_A = ["system.ai.gpt-5-6-sol", "system.ai.gpt-5-4-nano"] CODEX_B = ["system.ai.gpt-5-4-nano"] -# Model Provider Services published on the managed e2e workspace for these transitions. -CLAUDE_MPS = "main.default.ci_e2e_anthropic_mps" -CODEX_MPS = "main.default.ci_e2e_openai_mps" def _configure_managed(session, workspace): @@ -66,7 +58,7 @@ def _codex_listed(session): @pytest.mark.managed_fixture @pytest.mark.claude -def test_managed_fixture_claude_model_lifecycle(live_session, workspace, tmp_path): +def test_managed_fixture_claude_model_lifecycle(live_session, workspace): """Scenario: configure Claude across no config -> static A -> static B -> MPS -> no config, injecting each config (and an explicit null for the no-config states) via the stub. @@ -77,48 +69,33 @@ def test_managed_fixture_claude_model_lifecycle(live_session, workspace, tmp_pat """ session = live_session - set_managed_config_stub(session, tmp_path, None) + use_managed_config_fixture(session, "no_config") _configure_unmanaged(session, workspace, "claude") assert _claude_picker(session) is None - set_managed_config_stub( - session, - tmp_path, - build_coding_agent_config("CODING_AGENT_CLAUDE_CODE", build_claude_agent_config(CLAUDE_A)), - ) + use_managed_config_fixture(session, "claude_lifecycle_a") _configure_managed(session, workspace) assert _claude_picker(session) == CLAUDE_A - set_managed_config_stub( - session, - tmp_path, - build_coding_agent_config("CODING_AGENT_CLAUDE_CODE", build_claude_agent_config(CLAUDE_B)), - ) + use_managed_config_fixture(session, "claude_lifecycle_b") _configure_managed(session, workspace) assert _claude_picker(session) == CLAUDE_B assert "system.ai.claude-opus-4-8" not in _claude_picker(session) # Model discovery via MPS: the header routes, so the static picker is cleared. - set_managed_config_stub( - session, - tmp_path, - build_coding_agent_config( - "CODING_AGENT_CLAUDE_CODE", - build_mps_agent_config("CODING_AGENT_CLAUDE_CODE", CLAUDE_MPS), - ), - ) + use_managed_config_fixture(session, "claude_mps") _configure_managed(session, workspace) assert _claude_picker(session) is None # Config gone: the picker stays cleared, so an unmanaged workspace enforces no stale list. - set_managed_config_stub(session, tmp_path, None) + use_managed_config_fixture(session, "no_config") _configure_unmanaged(session, workspace, "claude") assert _claude_picker(session) is None @pytest.mark.managed_fixture @pytest.mark.codex -def test_managed_fixture_codex_model_lifecycle(live_session, workspace, tmp_path): +def test_managed_fixture_codex_model_lifecycle(live_session, workspace): """Scenario: configure Codex across no config -> static A -> static B -> MPS -> no config, injecting each config (and an explicit null for the no-config states) via the stub. @@ -130,39 +107,25 @@ def test_managed_fixture_codex_model_lifecycle(live_session, workspace, tmp_path """ session = live_session - set_managed_config_stub(session, tmp_path, None) + use_managed_config_fixture(session, "no_config") _configure_unmanaged(session, workspace, "codex") assert _codex_listed(session) == [] - set_managed_config_stub( - session, - tmp_path, - build_coding_agent_config("CODING_AGENT_CODEX", build_codex_agent_config(models=CODEX_A)), - ) + use_managed_config_fixture(session, "codex_lifecycle_a") _configure_managed(session, workspace) assert _codex_listed(session) == CODEX_A, _codex_listed(session) - set_managed_config_stub( - session, - tmp_path, - build_coding_agent_config("CODING_AGENT_CODEX", build_codex_agent_config(models=CODEX_B)), - ) + use_managed_config_fixture(session, "codex_lifecycle_b") _configure_managed(session, workspace) assert _codex_listed(session) == CODEX_B, _codex_listed(session) assert "system.ai.gpt-5-6-sol" not in _codex_listed(session) # Model discovery via MPS: the header routes, so the static catalog is cleared. - set_managed_config_stub( - session, - tmp_path, - build_coding_agent_config( - "CODING_AGENT_CODEX", build_mps_agent_config("CODING_AGENT_CODEX", CODEX_MPS) - ), - ) + use_managed_config_fixture(session, "codex_mps") _configure_managed(session, workspace) assert _codex_listed(session) == [] # Config gone: the catalog stays cleared, so an unmanaged workspace uses its own discovery. - set_managed_config_stub(session, tmp_path, None) + use_managed_config_fixture(session, "no_config") _configure_unmanaged(session, workspace, "codex") assert _codex_listed(session) == [] diff --git a/tests/integration/test_ug_configure_managed_mcp.py b/tests/integration/test_ug_configure_managed_mcp.py index b3d66f572..0b52643fc 100644 --- a/tests/integration/test_ug_configure_managed_mcp.py +++ b/tests/integration/test_ug_configure_managed_mcp.py @@ -1,8 +1,8 @@ """Managed-config CUJ: admin MCP servers reach the agent, additive to the developer's own servers. -The admin CodingAgentConfig is injected via UCODE_MANAGED_CONFIG_STUB so the real configure path -can be driven against an MCP list the live workspace does not publish; only the config INPUT is -stubbed (auth, the config writers, the OS-managed reconcile, and the agent binary stay real). See +The admin CodingAgentConfig is a checked-in JSON fixture injected via UCODE_MANAGED_CONFIG_STUB so +the real configure path can be driven against an MCP list; only the config INPUT is stubbed (auth, +the config writers, the OS-managed reconcile, and the agent binary stay real). See tests/AGENTS.md rule 4. An interactive `ug configure` (a PTY, so ug performs its sudo OS-managed reconcile) writes the @@ -16,37 +16,23 @@ import tomllib import pytest -from utils.managed import ( - build_claude_agent_config, - build_codex_agent_config, - build_coding_agent_config, - set_managed_config_stub, -) +from utils.managed import use_managed_config_fixture from utils.terminal import AgentTerminal, ConfigureTerminal -CLAUDE_OPUS = "system.ai.claude-opus-4-8" -CODEX_MODEL = "system.ai.gpt-5-6-sol" -# A real ca-central MCP service; the live published config lists no MCP servers, so its appearance -# can only come from the injected config. -MCP_SERVICE = "system.ai.github" +# The fixtures list `system.ai.github`, a real MCP service on the managed workspace. CODEX_MANAGED_CONFIG_PATH = "/etc/codex/managed_config.toml" @pytest.mark.managed_fixture @pytest.mark.claude -def test_managed_fixture_claude_mcp_lists_configured_server(live_session, workspace, tmp_path): +def test_managed_fixture_claude_mcp_lists_configured_server(live_session, workspace): """Scenario: a non-interactive configure with a managed MCP server, then open Claude's /mcp. Expected: with no sudo reconcile available, ug registers the server at user scope and the real /mcp view lists it. """ session = live_session - config = build_coding_agent_config( - "CODING_AGENT_CLAUDE_CODE", - build_claude_agent_config([CLAUDE_OPUS]), - mcp_names=[MCP_SERVICE], - ) - set_managed_config_stub(session, tmp_path, config) + use_managed_config_fixture(session, "claude_mcp") result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) assert "Select coding agents to configure:" not in result.stdout, result.stdout @@ -63,19 +49,14 @@ def test_managed_fixture_claude_mcp_lists_configured_server(live_session, worksp @pytest.mark.managed_fixture @pytest.mark.codex -def test_managed_fixture_codex_mcp_written_to_managed_file(live_session, workspace, tmp_path): +def test_managed_fixture_codex_mcp_written_to_managed_file(live_session, workspace): """Scenario: an interactive configure with a managed MCP server writes Codex's managed file. Expected: the server lands in the `[mcp_servers]` table of the OS-managed config as the ug mcp-proxy stdio command, and the developer's own ~/.codex/config.toml is not touched. """ session = live_session - config = build_coding_agent_config( - "CODING_AGENT_CODEX", - build_codex_agent_config(models=[CODEX_MODEL]), - mcp_names=[MCP_SERVICE], - ) - set_managed_config_stub(session, tmp_path, config) + use_managed_config_fixture(session, "codex_mcp") command = [str(session.binary), "configure", "--workspace", workspace, "--skip-upgrade"] with ConfigureTerminal(session, "codex", command, "managed-mcp-file-codex") as configure: configure.finish(timeout=300) @@ -97,7 +78,7 @@ def test_managed_fixture_codex_mcp_written_to_managed_file(live_session, workspa @pytest.mark.managed_fixture @pytest.mark.codex -def test_managed_fixture_codex_http_headers_in_managed_file(live_session, workspace, tmp_path): +def test_managed_fixture_codex_http_headers_in_managed_file(live_session, workspace): """Scenario: an interactive configure with managed http_headers writes them into the OS-managed file. Scope: managed-config content assertion in the same family as the MCP/skills/models @@ -113,14 +94,7 @@ def test_managed_fixture_codex_http_headers_in_managed_file(live_session, worksp session = live_session managed_header_key = "x-databricks-workspace" managed_header_value = "eng-ml-inference" - config = build_coding_agent_config( - "CODING_AGENT_CODEX", - build_codex_agent_config( - models=[CODEX_MODEL], - http_headers={managed_header_key: managed_header_value}, - ), - ) - set_managed_config_stub(session, tmp_path, config) + use_managed_config_fixture(session, "codex_http_headers") command = [str(session.binary), "configure", "--workspace", workspace, "--skip-upgrade"] with ConfigureTerminal(session, "codex", command, "managed-http-headers-codex") as configure: configure.finish(timeout=300) diff --git a/tests/integration/test_ug_configure_managed_models.py b/tests/integration/test_ug_configure_managed_models.py index f1ad52296..5dd34bb31 100644 --- a/tests/integration/test_ug_configure_managed_models.py +++ b/tests/integration/test_ug_configure_managed_models.py @@ -1,34 +1,21 @@ """Managed-config CUJs for agent model lists, smart routing, and Codex catalog fallback metadata. -The Claude defaults cases read published configs from their dedicated workspaces. Other cases -inject the admin CodingAgentConfig via UCODE_MANAGED_CONFIG_STUB to exercise shapes the live -workspace does not publish; only that config input is stubbed. See tests/AGENTS.md rule 4. +Every case injects a checked-in JSON CodingAgentConfig (tests/fixtures/managed_config/) via +UCODE_MANAGED_CONFIG_STUB on the managed workspace; only that config input is stubbed. See +tests/AGENTS.md rule 4. """ import json -import os import pytest -from utils.constants import ( - CLAUDE_SMART_ROUTING_MODELS, - CODEX_SMART_ROUTING_MODELS, - MANAGED_CLAUDE_PROVIDER_SERVICE, -) +from utils.constants import MANAGED_CLAUDE_PROVIDER_SERVICE from utils.evidence import FileTask -from utils.managed import ( - build_claude_agent_config, - build_codex_agent_config, - build_coding_agent_config, - set_managed_config_stub, -) +from utils.managed import use_managed_config_fixture from utils.provider_catalog import fetch_anthropic_parent_catalog, fetch_anthropic_provider_catalog from utils.terminal import AgentTerminal, TerminalProcess -CLAUDE_OPUS = "system.ai.claude-opus-4-8" -# A real ca-central model absent from the live published config: its presence in the picker can -# only come from the injected config, which the live workspace's model list cannot produce. -CLAUDE_OFF_MENU = "system.ai.claude-sonnet-5" -LIVE_ONLY = "haiku-4-5" # published live, but not in the injected list below +# The picker fixture lists sonnet-5 but not haiku-4-5, which the managed workspace's own list has. +LIVE_ONLY = "haiku-4-5" CODEX_DEFAULT = "system.ai.gpt-5-6-sol" CODEX_WITHOUT_BUNDLED_METADATA = "system.ai.gpt-99" MANAGED_CLAUDE_DEFAULT_ENV_KEYS = { @@ -37,20 +24,14 @@ "default_sonnet_model": "ANTHROPIC_DEFAULT_SONNET_MODEL", "default_haiku_model": "ANTHROPIC_DEFAULT_HAIKU_MODEL", } -CLAUDE_MPS_DEFAULTS_WORKSPACE = ( - "https://eng-ml-inference-batch-inference-us-west-2.cloud.databricks.com" -) -CLAUDE_PARENT_SCHEMA_DEFAULTS_WORKSPACE = ( - "https://eng-ml-inference-ap-northeast-2.cloud.databricks.com" -) SMART_ROUTING_BANNER = "Using Unity Gateway Smart Router." -@pytest.mark.managed +@pytest.mark.managed_fixture @pytest.mark.claude -def test_managed_claude_mps_defaults_accompany_discovery(live_session): - """Scenario: launch Claude with managed defaults and MPS discovery on the west-2 workspace. +def test_managed_claude_mps_defaults_accompany_discovery(live_session, workspace): + """Scenario: launch Claude under an injected config with defaults and MPS discovery. Expected: the installed ug launch writes the MPS header and every admin-authored default to both Claude settings files without changing the model ids, and replaces built-in picker rows @@ -59,9 +40,6 @@ def test_managed_claude_mps_defaults_accompany_discovery(live_session): model inference. """ session = live_session - target_bearer = os.environ.get("UG_MPS_DEFAULTS_BEARER", "").strip() - assert target_bearer, "The runner needs UG_MPS_DEFAULTS_CLIENT_SECRET for this workspace." - session.env["DATABRICKS_BEARER"] = target_bearer defaults = { "default_model": "anthropic.claude-sonnet-5", "default_fable_model": "anthropic.claude-fable-5-1", @@ -69,15 +47,12 @@ def test_managed_claude_mps_defaults_accompany_discovery(live_session): "default_sonnet_model": "anthropic.claude-sonnet-5", "default_haiku_model": "anthropic.claude-haiku-4-5", } - result = session.run( - "configure", "--workspace", CLAUDE_MPS_DEFAULTS_WORKSPACE, "--skip-upgrade", timeout=240 - ) + use_managed_config_fixture(session, "claude_mps_defaults") + result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) assert "Select coding agents to configure:" not in result.stdout, result.stdout catalog = fetch_anthropic_provider_catalog( - CLAUDE_MPS_DEFAULTS_WORKSPACE, - target_bearer, - MANAGED_CLAUDE_PROVIDER_SERVICE, + workspace, session.env["DATABRICKS_BEARER"], MANAGED_CLAUDE_PROVIDER_SERVICE ) expected_families = ("opus", "sonnet", "haiku", "fable") command = [str(session.binary), "claude", "--", "--version"] @@ -109,10 +84,10 @@ def test_managed_claude_mps_defaults_accompany_discovery(live_session): assert option["label"] == display_name, option -@pytest.mark.managed +@pytest.mark.managed_fixture @pytest.mark.claude -def test_managed_claude_parent_schema_defaults_accompany_discovery(live_session): - """Scenario: launch Claude with managed defaults and UC discovery on the northeast-2 workspace. +def test_managed_claude_parent_schema_defaults_accompany_discovery(live_session, workspace): + """Scenario: launch Claude under an injected config with defaults and `system.ai` UC discovery. Expected: the installed ug launch writes the parent-schema header and every admin-authored default to both Claude settings files, adding ``[1m]`` only to Opus and Sonnet family defaults. @@ -121,11 +96,6 @@ def test_managed_claude_parent_schema_defaults_accompany_discovery(live_session) model inference. """ session = live_session - target_bearer = os.environ.get("UG_PARENT_SCHEMA_DEFAULTS_BEARER", "").strip() - assert target_bearer, ( - "The runner needs UG_PARENT_SCHEMA_DEFAULTS_CLIENT_SECRET for this workspace." - ) - session.env["DATABRICKS_BEARER"] = target_bearer parent_schema = "system.ai" defaults = { "default_model": f"{parent_schema}.claude-sonnet-5", @@ -134,17 +104,12 @@ def test_managed_claude_parent_schema_defaults_accompany_discovery(live_session) "default_sonnet_model": f"{parent_schema}.claude-sonnet-5", "default_haiku_model": f"{parent_schema}.claude-haiku-4-5", } - result = session.run( - "configure", - "--workspace", - CLAUDE_PARENT_SCHEMA_DEFAULTS_WORKSPACE, - "--skip-upgrade", - timeout=240, - ) + use_managed_config_fixture(session, "claude_parent_schema_defaults") + result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) assert "Select coding agents to configure:" not in result.stdout, result.stdout catalog = fetch_anthropic_parent_catalog( - CLAUDE_PARENT_SCHEMA_DEFAULTS_WORKSPACE, target_bearer, parent_schema + workspace, session.env["DATABRICKS_BEARER"], parent_schema ) command = [str(session.binary), "claude", "--", "--version"] with TerminalProcess(session, "claude", command, "managed-defaults-parent-schema") as terminal: @@ -183,17 +148,14 @@ def test_managed_claude_parent_schema_defaults_accompany_discovery(live_session) @pytest.mark.managed_fixture @pytest.mark.claude -def test_managed_fixture_claude_model_picker_reflects_the_config(live_session, workspace, tmp_path): +def test_managed_fixture_claude_model_picker_reflects_the_config(live_session, workspace): """Scenario: launch Claude under an injected managed config and open the /model picker. Expected: the picker offers the injected models (including one the live workspace does not publish) and omits a model the live workspace does publish. """ session = live_session - config = build_coding_agent_config( - "CODING_AGENT_CLAUDE_CODE", build_claude_agent_config([CLAUDE_OPUS, CLAUDE_OFF_MENU]) - ) - set_managed_config_stub(session, tmp_path, config) + use_managed_config_fixture(session, "claude_model_picker") result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) assert "Select coding agents to configure:" not in result.stdout, result.stdout @@ -210,7 +172,7 @@ def test_managed_fixture_claude_model_picker_reflects_the_config(live_session, w @pytest.mark.managed_fixture @pytest.mark.codex @pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_ug_configure_managed_codex_catalog_fallback(live_session, workspace, tmp_path, routing): +def test_ug_configure_managed_codex_catalog_fallback(live_session, workspace, routing): """Scenario: configure Codex from an injected model list containing an unknown GPT model. Expected: ug creates conservative fallback metadata for the unknown model, warns how to get @@ -218,11 +180,7 @@ def test_ug_configure_managed_codex_catalog_fallback(live_session, workspace, tm without smart routing. """ session = live_session - config = build_coding_agent_config( - "CODING_AGENT_CODEX", - build_codex_agent_config(models=[CODEX_DEFAULT, CODEX_WITHOUT_BUNDLED_METADATA]), - ) - set_managed_config_stub(session, tmp_path, config) + use_managed_config_fixture(session, "codex_catalog_fallback") result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) assert "Select coding agents to configure:" not in result.stdout, result.stdout assert "Codex is missing metadata for managed GPT model" in result.stdout, result.stdout @@ -259,7 +217,7 @@ def test_ug_configure_managed_codex_catalog_fallback(live_session, workspace, tm @pytest.mark.managed_fixture @pytest.mark.claude -def test_managed_fixture_claude_smart_routing_banner(live_session, workspace, tmp_path): +def test_managed_fixture_claude_smart_routing_banner(live_session, workspace): """Scenario: an admin config lists Claude models and enables smart routing for Claude. Expected: `ug configure` applies the config without the personal agent selector, and @@ -269,11 +227,7 @@ def test_managed_fixture_claude_smart_routing_banner(live_session, workspace, tm """ session = live_session task = FileTask(session) - config = build_coding_agent_config( - "CODING_AGENT_CLAUDE_CODE", - build_claude_agent_config(CLAUDE_SMART_ROUTING_MODELS, smart_routing=True), - ) - set_managed_config_stub(session, tmp_path, config) + use_managed_config_fixture(session, "claude_smart_routing") result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) assert "Select coding agents to configure:" not in result.stdout, result.stdout @@ -295,7 +249,7 @@ def test_managed_fixture_claude_smart_routing_banner(live_session, workspace, tm @pytest.mark.managed_fixture @pytest.mark.codex -def test_managed_fixture_codex_smart_routing_banner(live_session, workspace, tmp_path): +def test_managed_fixture_codex_smart_routing_banner(live_session, workspace): """Scenario: an admin config lists Codex models and enables smart routing for Codex. Expected: `ug configure` applies the config without the personal agent selector, and @@ -305,11 +259,7 @@ def test_managed_fixture_codex_smart_routing_banner(live_session, workspace, tmp """ session = live_session task = FileTask(session) - config = build_coding_agent_config( - "CODING_AGENT_CODEX", - build_codex_agent_config(models=CODEX_SMART_ROUTING_MODELS, smart_routing=True), - ) - set_managed_config_stub(session, tmp_path, config) + use_managed_config_fixture(session, "codex_smart_routing") result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) assert "Select coding agents to configure:" not in result.stdout, result.stdout diff --git a/tests/integration/test_ug_configure_managed_skills.py b/tests/integration/test_ug_configure_managed_skills.py index 1a46b6b09..9a96eda57 100644 --- a/tests/integration/test_ug_configure_managed_skills.py +++ b/tests/integration/test_ug_configure_managed_skills.py @@ -1,8 +1,7 @@ """Managed-config CUJ: the launched agent's /skills view lists the admin's downloaded skills. -The admin CodingAgentConfig is injected via UCODE_MANAGED_CONFIG_STUB so the real /skills TUI can be -driven against a skill the live workspace's published config does not include; only the config INPUT -is stubbed (auth, the skill download, the config writers, and the agent binary stay real). These +The admin CodingAgentConfig is a checked-in JSON fixture injected via UCODE_MANAGED_CONFIG_STUB so +the real /skills TUI can be driven against a managed skill; only the config INPUT is stubbed (auth, the skill download, the config writers, and the agent binary stay real). These assert what the agent presents, not the bundles on disk (that is unit tests' job). See tests/AGENTS.md rule 4. Claude reads ~/.claude/skills and Codex reads the shared ~/.agents/skills, both of which `ug configure` writes, so a managed skill reaches either agent. The reconcile @@ -11,23 +10,13 @@ """ import pytest -from utils.managed import ( - build_claude_agent_config, - build_codex_agent_config, - build_coding_agent_config, - set_managed_config_stub, -) +from utils.managed import use_managed_config_fixture from utils.terminal import AgentTerminal -CLAUDE_OPUS = "system.ai.claude-opus-4-8" -CODEX_MODEL = "system.ai.gpt-5-6-sol" -# A real finalized skill in the managed e2e workspace's `main.default`. `ug configure` downloads it -# to disk, so its appearance in the agent's /skills view can only come from the injected config. -# `main.default.forkable-meals` names the securable; `forkable-meals` is the bundle name the agent -# lists (they match for this skill). Update if the workspace's skills change. -SKILL_FQN = "main.default.forkable-meals" +# A real finalized skill in the managed e2e workspace's `main.default`, named by the skills fixtures +# (`main.default.forkable-meals`); `forkable-meals` is the bundle name the agent lists (they match +# for this skill). Update the fixtures and this name if the workspace's skills change. SKILL_NAME = "forkable-meals" -SKILLS_LOCATION = "main.default" SKILL_ROOTS = (".claude/skills", ".agents/skills") # Bundle name for a hand-authored developer skill used in the reconcile lifecycle; it has no # attribution record, so it stands in for any skill ug did not download. @@ -42,19 +31,14 @@ def _bundles_on_disk(session) -> set[str]: @pytest.mark.managed_fixture @pytest.mark.claude -def test_managed_fixture_claude_skills_lists_downloaded_skill(live_session, workspace, tmp_path): +def test_managed_fixture_claude_skills_lists_downloaded_skill(live_session, workspace): """Scenario: configure Claude under an injected config that names a managed skill, open /skills. Expected: the downloaded managed skill (named via the `skills.names` selector) appears in the agent's /skills view. """ session = live_session - config = build_coding_agent_config( - "CODING_AGENT_CLAUDE_CODE", - build_claude_agent_config([CLAUDE_OPUS]), - skill_names=[SKILL_FQN], - ) - set_managed_config_stub(session, tmp_path, config) + use_managed_config_fixture(session, "claude_skills_by_name") result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) assert "Select coding agents to configure:" not in result.stdout, result.stdout @@ -71,7 +55,7 @@ def test_managed_fixture_claude_skills_lists_downloaded_skill(live_session, work @pytest.mark.managed_fixture @pytest.mark.codex -def test_managed_fixture_codex_skills_lists_downloaded_skill(live_session, workspace, tmp_path): +def test_managed_fixture_codex_skills_lists_downloaded_skill(live_session, workspace): """Scenario: configure Codex under an injected config with a managed skills schema, open /skills. Expected: the downloaded managed skill (from the `skills.unity_catalog_location` selector) @@ -79,12 +63,7 @@ def test_managed_fixture_codex_skills_lists_downloaded_skill(live_session, works writes, so the admin's skill reaches it too. """ session = live_session - config = build_coding_agent_config( - "CODING_AGENT_CODEX", - build_codex_agent_config(models=[CODEX_MODEL]), - skills_location=SKILLS_LOCATION, - ) - set_managed_config_stub(session, tmp_path, config) + use_managed_config_fixture(session, "codex_skills_by_location") result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=240) assert "Select coding agents to configure:" not in result.stdout, result.stdout @@ -102,7 +81,7 @@ def test_managed_fixture_codex_skills_lists_downloaded_skill(live_session, works @pytest.mark.managed_fixture @pytest.mark.claude -def test_managed_skills_reconcile_lifecycle(live_session, workspace, tmp_path): +def test_managed_skills_reconcile_lifecycle(live_session, workspace): """Scenario: managed skills download, coexist with a developer's own skill, then reconcile away. Expected: a `unity_catalog_location` config downloads the workspace's skills to disk; a @@ -111,21 +90,16 @@ def test_managed_skills_reconcile_lifecycle(live_session, workspace, tmp_path): proves the reconcile is attribution-driven: it never deletes a skill ug did not download. """ session = live_session - claude = build_claude_agent_config([CLAUDE_OPUS]) - def configure(config: dict) -> None: - set_managed_config_stub(session, tmp_path, config) + def configure(fixture: str) -> None: + use_managed_config_fixture(session, fixture) result = session.run("configure", "--workspace", workspace, "--skip-upgrade", timeout=300) assert "Select coding agents to configure:" not in result.stdout, result.stdout - configure(build_coding_agent_config("CODING_AGENT_CLAUDE_CODE", claude)) + configure("claude_no_skills") assert _bundles_on_disk(session) == set(), "a no-skills config must download nothing" - configure( - build_coding_agent_config( - "CODING_AGENT_CLAUDE_CODE", claude, skills_location=SKILLS_LOCATION - ) - ) + configure("claude_skills_by_location") managed_bundles = _bundles_on_disk(session) assert managed_bundles, "the managed skills should have been downloaded" for root in SKILL_ROOTS: @@ -140,7 +114,7 @@ def configure(config: dict) -> None: f"---\nname: {DEVELOPER_SKILL}\ndescription: a developer's own skill.\n---\n" ) - configure(build_coding_agent_config("CODING_AGENT_CLAUDE_CODE", claude)) + configure("claude_no_skills") remaining = _bundles_on_disk(session) assert remaining == {DEVELOPER_SKILL}, remaining assert not (managed_bundles & remaining), "managed skills should have been reconciled away" diff --git a/tests/integration/test_ug_smart_routing_hooks.py b/tests/integration/test_ug_smart_routing_hooks.py index 58ddcffe3..1d8318c30 100644 --- a/tests/integration/test_ug_smart_routing_hooks.py +++ b/tests/integration/test_ug_smart_routing_hooks.py @@ -11,19 +11,13 @@ import json import pytest -from utils.constants import CLAUDE_SMART_ROUTING_MODELS, CODEX_SMART_ROUTING_MODELS from utils.evidence import ( SubagentCalculation, assert_subagent_routed, assistant_answer_contains, read_jsonl, ) -from utils.managed import ( - build_claude_agent_config, - build_codex_agent_config, - build_coding_agent_config, - set_managed_config_stub, -) +from utils.managed import use_managed_config_fixture from utils.terminal import AgentTerminal SMART_ROUTING_BANNER = "Using Unity Gateway Smart Router." @@ -250,7 +244,7 @@ def test_smart_routing_codex_route_subagent_hook(live_session, workspace): @pytest.mark.live @pytest.mark.claude @pytest.mark.managed_fixture -def test_smart_router_skill_toggles_claude_subagent_routing(live_session, workspace, tmp_path): +def test_smart_router_skill_toggles_claude_subagent_routing(live_session, workspace): """Scenario: launch Claude with subagent routing enabled, spawn a child, invoke the installed Smart Router skill to turn routing off, spawn another child, turn routing back on through the skill, and spawn a third child in the same real TUI session. @@ -263,11 +257,7 @@ def test_smart_router_skill_toggles_claude_subagent_routing(live_session, worksp session = live_session session.env["ENABLE_SMART_ROUTING_V2"] = "1" session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" - config = build_coding_agent_config( - "CODING_AGENT_CLAUDE_CODE", - build_claude_agent_config(CLAUDE_SMART_ROUTING_MODELS, smart_routing=True), - ) - set_managed_config_stub(session, tmp_path, config) + use_managed_config_fixture(session, "claude_smart_routing") session.run( "configure", "--workspace", @@ -296,7 +286,7 @@ def test_smart_router_skill_toggles_claude_subagent_routing(live_session, worksp @pytest.mark.live @pytest.mark.codex @pytest.mark.managed_fixture -def test_smart_router_skill_toggles_codex_subagent_routing(live_session, workspace, tmp_path): +def test_smart_router_skill_toggles_codex_subagent_routing(live_session, workspace): """Scenario: launch Codex with subagent routing enabled, spawn a child, invoke the installed Smart Router skill to turn routing off, spawn another child, turn routing back on through the skill, and spawn a third child in the same real TUI session. @@ -309,11 +299,7 @@ def test_smart_router_skill_toggles_codex_subagent_routing(live_session, workspa session = live_session session.env["ENABLE_SMART_ROUTING_V2"] = "1" session.env["ENABLE_SMART_ROUTING_SUBAGENT_ONLY"] = "1" - config = build_coding_agent_config( - "CODING_AGENT_CODEX", - build_codex_agent_config(models=CODEX_SMART_ROUTING_MODELS, smart_routing=True), - ) - set_managed_config_stub(session, tmp_path, config) + use_managed_config_fixture(session, "codex_smart_routing") session.run( "configure", "--workspace", diff --git a/tests/integration/utils/constants.py b/tests/integration/utils/constants.py index 650669b72..57cfab96c 100644 --- a/tests/integration/utils/constants.py +++ b/tests/integration/utils/constants.py @@ -3,16 +3,5 @@ CLAUDE_TEST_MODEL = "system.ai.claude-haiku-4-5" CODEX_TEST_MODEL = "system.ai.gpt-5-4-nano" -CLAUDE_SMART_ROUTING_MODELS = [ - "system.ai.claude-sonnet-5", - "system.ai.claude-haiku-4-5", - "system.ai.claude-opus-4-8", -] -CODEX_SMART_ROUTING_MODELS = [ - "system.ai.gpt-5-6-sol", - "system.ai.gpt-5-6-terra", - "system.ai.gpt-5-6-luna", -] - MANAGED_CLAUDE_PROVIDER_SERVICE = "main.default.ci_e2e_anthropic_mps" MANAGED_CODEX_PROVIDER_SERVICE = "main.default.ci_e2e_openai_mps" diff --git a/tests/integration/utils/managed.py b/tests/integration/utils/managed.py index 3599d38f9..2eec1ac78 100644 --- a/tests/integration/utils/managed.py +++ b/tests/integration/utils/managed.py @@ -1,16 +1,24 @@ -"""Shared fetching, builders, and stub injection for managed-config integration suites. +"""Shared fixture loading and stub injection for managed-config integration suites. -Pure CodingAgentConfig construction plus the ``UCODE_MANAGED_CONFIG_STUB`` filesystem/env mechanics. +Checked-in CodingAgentConfig JSON fixtures plus the ``UCODE_MANAGED_CONFIG_STUB`` env mechanics. The configure invocation, launch, and assertions stay visible in each test (tests/AGENTS.md), and this module imports nothing from the ``ucode`` application package (tests/test_integration_contract.py enforces that boundary). """ -import json -import urllib.request from pathlib import Path MANAGED_CONFIGS_PATH = "/api/ai-gateway/v2/coding-agent-configs" +# One JSON CodingAgentConfig per scenario, in the GET shape `ug configure --file` will accept. +MANAGED_CONFIG_FIXTURES = Path(__file__).resolve().parents[2] / "fixtures" / "managed_config" + + +def use_managed_config_fixture(session, name: str) -> Path: + """Point ``UCODE_MANAGED_CONFIG_STUB`` at a checked-in fixture for this session.""" + path = MANAGED_CONFIG_FIXTURES / f"{name}.json" + assert path.is_file(), f"missing managed-config fixture {path}" + session.env["UCODE_MANAGED_CONFIG_STUB"] = str(path) + return path def assert_no_managed_config(payload: object) -> None: @@ -27,131 +35,6 @@ def assert_no_managed_config(payload: object) -> None: ) -def fetch_managed_config_stub( - workspace: str, - token: str, - directory: Path, - filename: str, - *, - agent: str, - provider_service: str, -) -> Path: - """Fetch the workspace config and persist one agent's MPS-backed test variant.""" - request = urllib.request.Request( - workspace.rstrip("/") + MANAGED_CONFIGS_PATH, - headers={"Authorization": f"Bearer {token}", "Accept": "application/json"}, - ) - with urllib.request.urlopen(request, timeout=30) as response: - payload = json.load(response) - - if isinstance(payload, dict): - configs = payload.get("coding_agent_configs") - elif isinstance(payload, list): - configs = payload - else: - configs = None - assert isinstance(configs, list) and configs, "workspace returned no managed CodingAgentConfig" - config = json.loads(json.dumps(configs[0])) - assert isinstance(config, dict), "managed CodingAgentConfig was not an object" - - enabled_agents = config.get("enabled_agents") - assert isinstance(enabled_agents, list), "managed CodingAgentConfig had no enabled_agents list" - matching_agents = [ - entry for entry in enabled_agents if isinstance(entry, dict) and entry.get("agent") == agent - ] - assert len(matching_agents) == 1, f"expected exactly one {agent} entry" - agent_config = matching_agents[0].get("config") - assert isinstance(agent_config, dict), f"{agent} had no managed agent config" - - agent_config["models"] = {"model_provider_service": provider_service} - agent_config.pop("default_models", None) - - stub = directory / filename - stub.write_text(json.dumps(config), encoding="utf-8") - stub.chmod(0o600) - return stub - - -def use_managed_config_stub(session, stub: Path) -> None: - """Point one isolated session at an already-persisted raw CodingAgentConfig.""" - session.env["UCODE_MANAGED_CONFIG_STUB"] = str(stub) - - -def set_managed_config_stub(session, tmp_path, config: dict | None) -> None: - """Write ``config`` to a file and point ``UCODE_MANAGED_CONFIG_STUB`` at it for this session. - - ``None`` writes an explicit JSON ``null``, which reproduces a workspace that publishes no - managed config (the stub's no-config state).""" - stub = Path(tmp_path) / "managed-config.json" - stub.write_text(json.dumps(config)) - use_managed_config_stub(session, stub) - - def is_managed_config_control_plane_cache(home: Path, path: Path) -> bool: """Whether ``path`` is ug's expected fetched-config cache, not agent-owned state.""" return path == home / ".ucode" / "managed-config.json" - - -def build_coding_agent_config( - default_agent: str, - *agents: dict, - mcp_names: list[str] | None = None, - skill_names: list[str] | None = None, - skills_location: str | None = None, -) -> dict: - config = {"spec_version": 1, "default_agent": default_agent, "enabled_agents": list(agents)} - if mcp_names is not None: - config["mcp_servers"] = {"names": mcp_names} - if skill_names is not None: - config["skills"] = {"names": skill_names} - elif skills_location is not None: - config["skills"] = {"unity_catalog_location": skills_location} - return config - - -def build_claude_agent_config( - models: list[str], - *, - family_defaults: dict[str, str] | None = None, - smart_routing: bool = False, - otel_tracing_enabled: bool | None = None, -) -> dict: - default_models = {"default_model": models[0]} - if family_defaults: - default_models.update( - {f"default_{family}_model": m for family, m in family_defaults.items()} - ) - config = { - "models": {"model_services": models}, - "default_models": default_models, - } - if smart_routing: - config["smart_routing"] = {"enabled": True} - if otel_tracing_enabled is not None: - config["tracing"] = {"enabled": otel_tracing_enabled} - return {"agent": "CODING_AGENT_CLAUDE_CODE", "config": config} - - -def build_codex_agent_config( - *, - models: list[str], - smart_routing: bool = False, - http_headers: dict[str, str] | None = None, - otel_tracing_enabled: bool | None = None, -) -> dict: - config = { - "models": {"model_services": models}, - "default_models": {"default_model": models[0]}, - } - if smart_routing: - config["smart_routing"] = {"enabled": True} - if http_headers is not None: - config["http_headers"] = http_headers - if otel_tracing_enabled is not None: - config["tracing"] = {"enabled": otel_tracing_enabled} - return {"agent": "CODING_AGENT_CODEX", "config": config} - - -def build_mps_agent_config(agent: str, provider: str) -> dict: - """A model-discovery config routing ``agent`` through a Model Provider Service (no static list).""" - return {"agent": agent, "config": {"models": {"model_provider_service": provider}}} diff --git a/tests/test_integration_contract.py b/tests/test_integration_contract.py index 07bd4577b..10af6ef48 100644 --- a/tests/test_integration_contract.py +++ b/tests/test_integration_contract.py @@ -161,7 +161,7 @@ def test_live_integration_cases_belong_to_exactly_one_ci_agent(): for node in tree.body: if isinstance(node, ast.FunctionDef) and node.name.startswith("test_"): marks = module_marks | _markers(node.decorator_list) - if marks & {"live", "managed", "workspace_switch"}: + if marks & {"live", "managed_fixture", "workspace_switch"}: assert len(marks & {"claude", "codex", "opencode"}) == 1, node.name @@ -198,7 +198,7 @@ def test_model_discovery_cases_match_current_launch_contract(): seen.append(case) marks = module_marks | _markers(node.decorator_list) expected = {"managed_fixture"} if case <= 6 else {"live"} - assert marks & {"managed_fixture", "managed", "live"} == expected, node.name + assert marks & {"managed_fixture", "live"} == expected, node.name assert marks & {"claude", "codex"} == ({"claude"} if case % 2 else {"codex"}), node.name assert not any(arg.arg == "configured" for arg in node.args.args), node.name for value in ast.walk(node):