Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 1 addition & 3 deletions .github/workflows/integration.yml
Original file line number Diff line number Diff line change
Expand Up @@ -392,8 +392,6 @@ jobs:
UCODE_TEST_WORKSPACE: ${{ secrets.E2E_ADMIN_WORKSPACE }}
DATABRICKS_CLIENT_ID: ${{ secrets.E2E_ADMIN_SP_CLIENT_ID }}
DATABRICKS_CLIENT_SECRET: ${{ secrets.E2E_ADMIN_SP_CLIENT_SECRET }}
UG_MPS_DEFAULTS_CLIENT_SECRET: ${{ matrix.agent == 'claude' && secrets.UG_MPS_DEFAULTS_CLIENT_SECRET || '' }}
UG_PARENT_SCHEMA_DEFAULTS_CLIENT_SECRET: ${{ matrix.agent == 'claude' && secrets.UG_PARENT_SCHEMA_DEFAULTS_CLIENT_SECRET || '' }}
UCODE_TEST_SECOND_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }}
DATABRICKS_SECOND_BEARER: ${{ secrets.DATABRICKS_BEARER }}
run: |
Expand All @@ -406,7 +404,7 @@ jobs:
uv run --no-project --python 3.12 python scripts/run_integration.py \
--python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \
--default-index "$PACKAGE_INDEX" --output "$RUNNER_TEMP/ug-integration" \
"${args[@]}" -- -m "(managed or managed_fixture or workspace_switch) and $AGENT"
"${args[@]}" -- -m "(managed_fixture or workspace_switch) and $AGENT"
- name: Upload managed test evidence
if: ${{ !cancelled() }}
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
Expand Down
34 changes: 1 addition & 33 deletions scripts/run_integration.py
Original file line number Diff line number Diff line change
Expand Up @@ -31,20 +31,6 @@
"codex": "@openai/codex",
"opencode": "opencode-ai",
}
MANAGED_DEFAULTS_TARGETS = (
(
"UG_MPS_DEFAULTS_BEARER",
"https://eng-ml-inference-batch-inference-us-west-2.cloud.databricks.com",
"1c359c0f-58bc-42ac-a74f-079ccb173676",
"UG_MPS_DEFAULTS_CLIENT_SECRET",
),
(
"UG_PARENT_SCHEMA_DEFAULTS_BEARER",
"https://eng-ml-inference-ap-northeast-2.cloud.databricks.com",
"95e267dc-4393-4360-9d45-4b9b13b2d370",
"UG_PARENT_SCHEMA_DEFAULTS_CLIENT_SECRET",
),
)
WINDOWS_PATHEXT = ".COM;.EXE;.BAT;.CMD"
UV_INDEX_CREDENTIAL_ENV = (
"UV_INDEX_DATABRICKS_PYPI_USERNAME",
Expand Down Expand Up @@ -520,12 +506,7 @@ def terminate(signum, frame):
bearer = os.environ.get("DATABRICKS_BEARER", "").strip()
second_bearer = os.environ.get("DATABRICKS_SECOND_BEARER", "").strip()
oauth_token = os.environ.get("CLAUDE_CODE_OAUTH_TOKEN", "").strip()
target_bearers: dict[str, str] = {}
client_secrets = (
os.environ.get("DATABRICKS_CLIENT_SECRET", ""),
os.environ.get("UG_MPS_DEFAULTS_CLIENT_SECRET", ""),
os.environ.get("UG_PARENT_SCHEMA_DEFAULTS_CLIENT_SECRET", ""),
)
client_secrets = (os.environ.get("DATABRICKS_CLIENT_SECRET", ""),)

def redact(value: str) -> str:
return redact_secrets(
Expand All @@ -534,7 +515,6 @@ def redact(value: str) -> str:
bearer,
second_bearer,
oauth_token,
*target_bearers.values(),
*client_secrets,
*installer_secrets,
),
Expand Down Expand Up @@ -831,14 +811,6 @@ def run(command, *, cwd=output, env=base_env, timeout=600) -> str:
if client_id and client_secret:
bearer = mint_m2m_token(args.workspace, client_id, client_secret)

if not args.installation_only:
for bearer_env, target_workspace, client_id, secret_env in MANAGED_DEFAULTS_TARGETS:
secret = os.environ.get(secret_env, "").strip()
if args.workspace.rstrip("/") == target_workspace:
target_bearers[bearer_env] = bearer
elif secret:
target_bearers[bearer_env] = mint_m2m_token(target_workspace, client_id, secret)

test_dependencies = ["pytest==9.0.3"]
if os.name == "posix":
test_dependencies.extend(["pexpect==4.9.0", "pyte==0.8.2"])
Expand Down Expand Up @@ -876,10 +848,6 @@ def run(command, *, cwd=output, env=base_env, timeout=600) -> str:
"UG_INTEGRATION_CODEX_PARENT_MODEL": args.codex_parent_model,
"UCODE_TEST_WORKSPACE": args.workspace or "",
"DATABRICKS_BEARER": bearer,
"UG_MPS_DEFAULTS_BEARER": target_bearers.get("UG_MPS_DEFAULTS_BEARER", ""),
"UG_PARENT_SCHEMA_DEFAULTS_BEARER": target_bearers.get(
"UG_PARENT_SCHEMA_DEFAULTS_BEARER", ""
),
"UCODE_TEST_SECOND_WORKSPACE": args.second_workspace or "",
"DATABRICKS_SECOND_BEARER": second_bearer,
"UG_INTEGRATION_WAREHOUSE_ID": args.warehouse_id or "",
Expand Down
11 changes: 6 additions & 5 deletions tests/AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -29,11 +29,12 @@ tests. Keep work scoped to the behavior requested by the user.
and regression coverage. Report failures instead of concealing them. The sole
exception is the `managed_fixture` marker: it uses the built-in
`UCODE_MANAGED_CONFIG_STUB` hook to inject the admin CodingAgentConfig INPUT so the
real `ug configure` path can be exercised across config shapes the live workspace does
not publish and one workspace fetch can be replayed across isolated cases. The gateway,
agent binaries, ug internals, and ug state stay real; the config fetch/wire contract stays
covered by the un-stubbed `managed` tests; and the hook must never be used to disable
validation or conceal a failure.
real `ug configure` path runs against a known config on the common managed workspace.
Every injected config is a checked-in JSON file in `fixtures/managed_config/` (the GET
shape of one CodingAgentConfig), so `ug configure --file` (#866) can replace the hook
mechanically. The gateway, agent binaries, ug internals, and ug state stay real; the
config fetch/wire contract is covered by unit tests and the read-only `e2e_cuj/` CUJs;
and the hook must never be used to disable validation or conceal a failure.
5. **Real responses and binaries.** Pin requested ug and agent versions. Never
substitute a missing binary/service. Reuse explicit e2e workspace/auth settings;
never pick a developer's Databricks profile automatically.
Expand Down
15 changes: 7 additions & 8 deletions tests/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -184,15 +184,15 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi
| `test_ug_configure_claude_repeat_and_revert`, `test_ug_configure_codex_repeat_and_revert` | Configure twice over user settings; complete a task; revert twice | Settings preserved; no bearer in ug state; generated config removed; status unconfigured |
| `test_ug_configure_claude_cleans_stale_skills_mcp_on_workspace_switch` | Configure the first workspace, register its skills MCP, switch to a second real workspace, and use Claude | Old registration removed from Claude and the new workspace state; old workspace bucket preserved; repeat configure stays clean; real file task completes on the second workspace |
| `test_ug_configure_claude_rejects_invalid_credentials`, `test_ug_configure_codex_rejects_invalid_credentials` | Configure with a rejected bearer against the real workspace | Authentication failure; no successful saved setup |
| `test_ug_configure_managed_claude`, `test_ug_configure_managed_codex` | Configure against a workspace that publishes a managed CodingAgentConfig | No agent selector; each agent's generated config exposes exactly the admin's static model_services; real gateway prompt on launch. The Codex case also checks the shared catalog pointer, restart guidance, and a fresh bare app-server's visible model list |
| `test_ug_configure_managed_claude`, `test_ug_configure_managed_codex` | Configure on the managed workspace with the stubbed `managed_workspace_default.json` CodingAgentConfig | No agent selector; each agent's generated config exposes exactly the admin's static model_services; real gateway prompt on launch. The Codex case also checks the shared catalog pointer, restart guidance, and a fresh bare app-server's visible model list |
| `e2e_cuj/test_ug_budget_defaults.py` | Launch bare `ug` with separate low-spend and above-tier principals against the fixed 1% tier (`spending_percentage=0.01`); explicitly launch `ug claude` above the tier | Bare launches check Claude/Sonnet below the tier and the Codex/Luna recommendation above it; `ug usage` agrees with backend spend, threshold, and percentage. Explicit `ug claude` displays the backend's Codex/Luna recommendation while its generated model setting and native header select Sonnet. No budget writes or inference tasks. Run in the shared `E2E CUJs` job with both credential pairs documented in `integration/README.md`. |
| `test_case_01_*` | Launch managed Claude without defaults after configure and from fresh state | Claude receives the admin MPS header; its gateway cache and replacement picker match the independently fetched provider model IDs; catalog labels are preserved and a model appears in a numbered picker row |
| `test_case_03_*`, `test_case_05_*` | Pass a provider or model-location override to managed Claude after configure and from fresh state | ug rejects the override before Claude starts and preserves agent-owned state |
| `test_case_02_*` | Launch managed Codex after configure and from fresh state | The scoped and stable catalogs, ug-launched app server, and fresh bare app server match the independently fetched admin MPS model IDs. The configured case uses real `ug revert` to remove ug's shared pointer and stable file while preserving a user setting |
| `test_case_04_*`, `test_case_06_*` | Pass a provider or model-location override to managed Codex after configure and from fresh state | ug rejects the override before Codex starts and preserves agent-owned state |
| `test_ug_configure_managed_codex_catalog_fallback` | Configure from an injected managed response containing a GPT model absent from Codex's bundled catalog | Actionable metadata warning; conservative catalog entry for the unknown model; real Codex prompt on the valid default model |
| `test_managed_fixture_codex_http_headers_in_managed_file` | Interactive PTY configure with injected managed `http_headers` for Codex | The specified header (`x-databricks-workspace`) lands in `model_providers.Databricks.http_headers` in `/etc/codex/managed_config.toml` with the exact admin value |
| `test_managed_claude_mps_defaults_accompany_discovery`, `test_managed_claude_parent_schema_defaults_accompany_discovery` | Configure from the published admin config and launch Claude with MPS on `eng-ml-inference-batch-inference-us-west-2` and Unity Catalog discovery on `eng-ml-inference-ap-northeast-2`, respectively | Both generated settings files retain every admin-authored default alongside the source header and every independently fetched catalog model with its label; MPS pickers keep family shortcut rows separate from catalog entries; only UC Opus/Sonnet family ids gain `[1m]` |
| `test_managed_claude_mps_defaults_accompany_discovery`, `test_managed_claude_parent_schema_defaults_accompany_discovery` | Configure from a stubbed config and launch Claude with MPS discovery (`main.default.ci_e2e_anthropic_mps`) and with `system.ai` Unity Catalog discovery, respectively, both on the managed workspace | Both generated settings files retain every admin-authored default alongside the source header and every independently fetched catalog model with its label; MPS pickers keep family shortcut rows separate from catalog entries; only UC Opus/Sonnet family ids gain `[1m]` |
| `test_unmanaged_claude_preserves_preexisting_family_defaults` | Seed Claude's OS-managed family defaults, then configure against one real workspace verified to have no managed config | Every pre-existing Claude family default remains unchanged in the OS-managed settings file |
| `test_managed_fixture_claude_model_lifecycle`, `test_managed_fixture_codex_model_lifecycle` | Configure across no config -> static A -> static B -> MPS -> no config (stub-injected, `null` for no-config; MPS via a real provider service) | Each agent's model files reconcile to each static config (removed models pruned); switching to an MPS and a workspace with no managed config clears ug's static picker/catalog so no stale list is enforced |
| `test_ug_installed_wheel_exposes_help_and_version` | Invoke freshly installed console command | Package version matches; public help works |
Expand All @@ -202,13 +202,12 @@ integration utilities; only CUJ-specific evidence correlation stays in a test fi
| `test_ug_and_ucode_web_search_helpers_preserve_mcp_stdio` | Initialize and list tools through both web-search helper commands | Exactly the MCP JSON-RPC responses; no text/ANSI contamination; existing server/tool identities preserved; no model request |

With Claude and Codex selected there are **62 live cases** (12 marked TUI cases),
**6 managed-workspace cases** (marker `managed`, run against workspaces that
publish a CodingAgentConfig), **1 two-workspace case** (marker `workspace_switch`),
**25 managed-fixture cases** (marker `managed_fixture`, with only
the CodingAgentConfig input injected), and **7 installation checks**. The 14 retained numbered scenarios
**1 two-workspace case** (marker `workspace_switch`),
**33 managed-fixture cases** (marker `managed_fixture`, with only
the CodingAgentConfig input injected from a JSON file in `fixtures/managed_config/`), and **7 installation checks**. The 14 retained numbered scenarios
comprise **24 explicit journeys**: 12 managed configured/fresh executions and 12 unmanaged
executions. Thirteen additional managed-fixture cases cover focused model, MCP, skills,
and lifecycle shapes; two published-config cases cover Claude defaults. Parametrization varies
executions. The remaining managed-fixture cases cover focused model, MCP, skills,
cache-TTL, and lifecycle shapes, including two Claude defaults cases. Parametrization varies
argument spelling or routing mode, never hides the agent/provider in the test name. Duplicate boot-only cases
are incorporated into the Databricks configuration TUI journeys.
Generated-file cleanup and strict app-server stdout assertions remain enforced.
Expand Down
21 changes: 21 additions & 0 deletions tests/fixtures/managed_config/claude_lifecycle_a.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,21 @@
{
"spec_version": 1,
"default_agent": "CODING_AGENT_CLAUDE_CODE",
"enabled_agents": [
{
"agent": "CODING_AGENT_CLAUDE_CODE",
"config": {
"models": {
"model_services": [
"system.ai.claude-opus-4-8",
"system.ai.claude-sonnet-4-6",
"system.ai.claude-haiku-4-5"
]
},
"default_models": {
"default_model": "system.ai.claude-opus-4-8"
}
}
}
]
}
20 changes: 20 additions & 0 deletions tests/fixtures/managed_config/claude_lifecycle_b.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
{
"spec_version": 1,
"default_agent": "CODING_AGENT_CLAUDE_CODE",
"enabled_agents": [
{
"agent": "CODING_AGENT_CLAUDE_CODE",
"config": {
"models": {
"model_services": [
"system.ai.claude-sonnet-4-6",
"system.ai.claude-haiku-4-5"
]
},
"default_models": {
"default_model": "system.ai.claude-sonnet-4-6"
}
}
}
]
}
24 changes: 24 additions & 0 deletions tests/fixtures/managed_config/claude_mcp.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
{
"spec_version": 1,
"default_agent": "CODING_AGENT_CLAUDE_CODE",
"enabled_agents": [
{
"agent": "CODING_AGENT_CLAUDE_CODE",
"config": {
"models": {
"model_services": [
"system.ai.claude-opus-4-8"
]
},
"default_models": {
"default_model": "system.ai.claude-opus-4-8"
}
}
}
],
"mcp_servers": {
"names": [
"system.ai.github"
]
}
}
20 changes: 20 additions & 0 deletions tests/fixtures/managed_config/claude_model_picker.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
{
"spec_version": 1,
"default_agent": "CODING_AGENT_CLAUDE_CODE",
"enabled_agents": [
{
"agent": "CODING_AGENT_CLAUDE_CODE",
"config": {
"models": {
"model_services": [
"system.ai.claude-opus-4-8",
"system.ai.claude-sonnet-5"
]
},
"default_models": {
"default_model": "system.ai.claude-opus-4-8"
}
}
}
]
}
14 changes: 14 additions & 0 deletions tests/fixtures/managed_config/claude_mps.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,14 @@
{
"spec_version": 1,
"default_agent": "CODING_AGENT_CLAUDE_CODE",
"enabled_agents": [
{
"agent": "CODING_AGENT_CLAUDE_CODE",
"config": {
"models": {
"model_provider_service": "main.default.ci_e2e_anthropic_mps"
}
}
}
]
}
21 changes: 21 additions & 0 deletions tests/fixtures/managed_config/claude_mps_defaults.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,21 @@
{
"spec_version": 1,
"default_agent": "CODING_AGENT_CLAUDE_CODE",
"enabled_agents": [
{
"agent": "CODING_AGENT_CLAUDE_CODE",
"config": {
"models": {
"model_provider_service": "main.default.ci_e2e_anthropic_mps"
},
"default_models": {
"default_model": "anthropic.claude-sonnet-5",
"default_fable_model": "anthropic.claude-fable-5-1",
"default_opus_model": "anthropic.claude-opus-5",
"default_sonnet_model": "anthropic.claude-sonnet-5",
"default_haiku_model": "anthropic.claude-haiku-4-5"
}
}
}
]
}
19 changes: 19 additions & 0 deletions tests/fixtures/managed_config/claude_no_skills.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
{
"spec_version": 1,
"default_agent": "CODING_AGENT_CLAUDE_CODE",
"enabled_agents": [
{
"agent": "CODING_AGENT_CLAUDE_CODE",
"config": {
"models": {
"model_services": [
"system.ai.claude-opus-4-8"
]
},
"default_models": {
"default_model": "system.ai.claude-opus-4-8"
}
}
}
]
}
21 changes: 21 additions & 0 deletions tests/fixtures/managed_config/claude_parent_schema_defaults.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,21 @@
{
"spec_version": 1,
"default_agent": "CODING_AGENT_CLAUDE_CODE",
"enabled_agents": [
{
"agent": "CODING_AGENT_CLAUDE_CODE",
"config": {
"models": {
"unity_catalog_location": "system.ai"
},
"default_models": {
"default_model": "system.ai.claude-sonnet-5",
"default_fable_model": "system.ai.claude-fable-5-1",
"default_opus_model": "system.ai.claude-opus-5",
"default_sonnet_model": "system.ai.claude-sonnet-5",
"default_haiku_model": "system.ai.claude-haiku-4-5"
}
}
}
]
}
22 changes: 22 additions & 0 deletions tests/fixtures/managed_config/claude_skills_by_location.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,22 @@
{
"spec_version": 1,
"default_agent": "CODING_AGENT_CLAUDE_CODE",
"enabled_agents": [
{
"agent": "CODING_AGENT_CLAUDE_CODE",
"config": {
"models": {
"model_services": [
"system.ai.claude-opus-4-8"
]
},
"default_models": {
"default_model": "system.ai.claude-opus-4-8"
}
}
}
],
"skills": {
"unity_catalog_location": "main.default"
}
}
24 changes: 24 additions & 0 deletions tests/fixtures/managed_config/claude_skills_by_name.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
{
"spec_version": 1,
"default_agent": "CODING_AGENT_CLAUDE_CODE",
"enabled_agents": [
{
"agent": "CODING_AGENT_CLAUDE_CODE",
"config": {
"models": {
"model_services": [
"system.ai.claude-opus-4-8"
]
},
"default_models": {
"default_model": "system.ai.claude-opus-4-8"
}
}
}
],
"skills": {
"names": [
"main.default.forkable-meals"
]
}
}
24 changes: 24 additions & 0 deletions tests/fixtures/managed_config/claude_smart_routing.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
{
"spec_version": 1,
"default_agent": "CODING_AGENT_CLAUDE_CODE",
"enabled_agents": [
{
"agent": "CODING_AGENT_CLAUDE_CODE",
"config": {
"models": {
"model_services": [
"system.ai.claude-sonnet-5",
"system.ai.claude-haiku-4-5",
"system.ai.claude-opus-4-8"
]
},
"default_models": {
"default_model": "system.ai.claude-sonnet-5"
},
"smart_routing": {
"enabled": true
}
}
}
]
}
Loading
Loading