From cdfe5f96eda3af59d7131555baf71c7e8168736d Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Mon, 14 Sep 2026 12:19:58 -0400 Subject: [PATCH 1/9] Run smoke and full journeys in parallel with clear CI checks --- .github/workflows/ci.yml | 69 ++++-- .github/workflows/integration.yml | 222 ++++++++++++++++++ tests/README.md | 20 +- tests/integration/README.md | 124 ++++++++++ tests/integration/pytest.ini | 1 + tests/integration/test_ug_claude_headless.py | 1 + tests/integration/test_ug_codex_headless.py | 1 + tests/integration/test_ug_configure_claude.py | 1 + tests/integration/test_ug_configure_codex.py | 1 + tests/test_integration_contract.py | 46 ++++ 10 files changed, 467 insertions(+), 19 deletions(-) create mode 100644 .github/workflows/integration.yml diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a86cc5913..f6e151e06 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -11,6 +11,7 @@ permissions: jobs: test: + name: Unit tests runs-on: ubuntu-latest env: # uv.lock pins packages to Databricks' internal pypi-proxy, which hosted @@ -23,8 +24,41 @@ jobs: - run: uv lock - run: uv run pytest --ignore=tests/test_e2e.py --ignore=tests/test_e2e_tracing.py - e2e: + e2e-shards: + name: ${{ matrix.name }} runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + include: + - group: gateway + name: Gateway API tests + keyword: not (TestCodexLaunch or TestClaudeLaunch or TestModelProviderLaunch or TestGeminiLaunch or TestGeminiFreshInstall or TestGeminiAuthRecovery or TestOpencodeLaunch or TestCopilotLaunch or TestPiLaunch) + package: '' + - group: claude + name: Agent launch tests · Claude + keyword: TestClaudeLaunch or (TestModelProviderLaunch and not codex) + package: '@anthropic-ai/claude-code' + - group: codex + name: Agent launch tests · Codex + keyword: TestCodexLaunch or (TestModelProviderLaunch and codex) + package: '@openai/codex' + - group: gemini + name: Agent launch tests · Gemini + keyword: TestGeminiLaunch or TestGeminiFreshInstall or TestGeminiAuthRecovery + package: '@google/gemini-cli' + - group: opencode + name: Agent launch tests · OpenCode + keyword: TestOpencodeLaunch + package: opencode-ai@1 + - group: copilot + name: Agent launch tests · Copilot + keyword: TestCopilotLaunch + package: '@github/copilot@1.0.80' + - group: pi + name: Agent launch tests · Pi + keyword: TestPiLaunch + package: '@earendil-works/pi-coding-agent' env: # See the test job: hosted runners can't reach the internal pypi-proxy, # so resolve against public PyPI (paired with the `uv lock` step below). @@ -42,34 +76,39 @@ jobs: DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} # Subscription OAuth token for the relayed-launch e2e test; the test skips when unset. CLAUDE_CODE_OAUTH_TOKEN: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }} + AGENT_PACKAGE: ${{ matrix.package }} + TEST_KEYWORD: ${{ matrix.keyword }} steps: - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 - uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0 - uses: databricks/setup-cli@bdb89f81c11a5bd647fd55b585b7c396ec68a25a # v1.0.0 - # The agent launch tests `_require_binary("codex")` etc. and skip when - # the CLI isn't on PATH. Install all six so each TestXxxLaunch test - # actually runs instead of skipping. - # Pin Copilot because 1.0.81-6 sends unsupported parameters through BYOK providers. - - name: Install agent CLIs - run: npm install -g - @anthropic-ai/claude-code - @openai/codex - @google/gemini-cli - opencode-ai@1 - @github/copilot@1.0.80 - @earendil-works/pi-coding-agent + - name: Install the shard's agent CLI + if: ${{ matrix.package != '' }} + run: npm install -g "$AGENT_PACKAGE" - run: uv lock - run: uv tool install . # Redirect stdin so any interactive `databricks auth login --no-browser` # fallback EOFs instead of hanging the runner. With DATABRICKS_BEARER # set, the auth code path doesn't shell out at all — this is a safety # net for any code path we may have missed. - - run: uv run pytest tests/test_e2e.py -v < /dev/null + - run: uv run pytest tests/test_e2e.py -v -k "$TEST_KEYWORD" < /dev/null # MLflow tracing e2e lives in its own file and needs the `tracing` # extra so `import mlflow` resolves (otherwise the test importorskips # and silently passes as skipped). `!cancelled()` lets this run even # when the previous pytest step failed — the two suites are # independent and one shouldn't mask the other. - - if: ${{ !cancelled() }} + - if: ${{ !cancelled() && matrix.group == 'claude' }} run: uv run --extra tracing pytest tests/test_e2e_tracing.py -v < /dev/null + + e2e: + name: All agent tests + if: ${{ !cancelled() }} + needs: e2e-shards + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - name: Require every e2e shard to pass + env: + RESULT: ${{ needs.e2e-shards.result }} + run: test "$RESULT" = success diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml new file mode 100644 index 000000000..e99e7645b --- /dev/null +++ b/.github/workflows/integration.yml @@ -0,0 +1,222 @@ +name: Integration + +on: + pull_request: + paths: ['src/**', 'tests/**', 'scripts/run_integration.py', 'pyproject.toml', 'uv.lock', '.github/workflows/integration.yml'] + push: + branches: [main] + paths: ['src/**', 'tests/**', 'scripts/run_integration.py', 'pyproject.toml', 'uv.lock', '.github/workflows/integration.yml'] + workflow_dispatch: + inputs: + suite: + description: Which integration checks to run + type: choice + options: [full, smoke, live, tui, installation] + default: full + ug_version: + description: Exact ucode release, or checkout + default: checkout + required: true + entry_point: + description: Console command (older releases may only have ucode) + type: choice + options: [ug, ucode] + default: ug + claude_version: + description: Exact Claude Code version + default: 2.1.268 + required: true + codex_version: + description: Exact Codex version + default: 0.154.0 + required: true + dependency: + description: Optional exact dependency for reproduction (e.g. tomlkit==0.14.0) + default: '' + required: false + index_url: + description: Python package index containing the requested ug version + default: https://pypi.org/simple + required: true + +permissions: + contents: read + +concurrency: + group: integration-${{ github.event.pull_request.number || github.ref }}-${{ github.event_name }} + cancel-in-progress: true + +env: + EXPECTED_WORKSPACE: https://eng-ml-inference-team-us-east-1.cloud.databricks.com + UG_VERSION: ${{ inputs.ug_version || 'checkout' }} + ENTRY_POINT: ${{ inputs.entry_point || 'ug' }} + CLAUDE_VERSION: ${{ inputs.claude_version || '2.1.268' }} + CODEX_VERSION: ${{ inputs.codex_version || '0.154.0' }} + PACKAGE_INDEX: ${{ inputs.index_url || 'https://pypi.org/simple' }} + +jobs: + installation: + name: Installation tests + runs-on: ubuntu-22.04 + timeout-minutes: 20 + steps: + - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + with: + fetch-depth: 0 + - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 + with: + node-version: 22.19.0 + - uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0 + with: + version: 0.9.8 + - name: Test a fresh installed package without credentials + run: | + uv run --no-project --python 3.12 python scripts/run_integration.py \ + --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ + --claude-version "$CLAUDE_VERSION" --codex-version "$CODEX_VERSION" \ + --index-url "$PACKAGE_INDEX" --installation-only --output "$RUNNER_TEMP/ug-integration" + - name: Upload installation evidence + if: ${{ !cancelled() }} + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 + with: + name: integration-installation + include-hidden-files: true + path: | + ${{ runner.temp }}/ug-integration/*.json + ${{ runner.temp }}/ug-integration/*.txt + ${{ runner.temp }}/ug-integration/*.xml + ${{ runner.temp }}/ug-integration/*.log + ${{ runner.temp }}/ug-integration/artifacts/ + ${{ runner.temp }}/ug-integration/wheels/ + + workspace: + name: Workspace validation + if: ${{ inputs.suite != 'installation' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) }} + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - name: Verify the existing e2e workspace and credential are configured + env: + UCODE_TEST_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} + DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} + run: | + python3 - <<'PY' + import os + expected = os.environ["EXPECTED_WORKSPACE"].rstrip("/") + actual = os.environ["UCODE_TEST_WORKSPACE"].strip().rstrip("/") + if actual != expected: + raise SystemExit("UCODE_TEST_WORKSPACE does not match the expected CI workspace.") + if not os.environ["DATABRICKS_BEARER"].strip(): + raise SystemExit("The existing e2e DATABRICKS_BEARER secret is missing.") + print("CI workspace matches; the existing e2e credential is present.") + PY + + smoke: + name: Smoke journeys · ${{ matrix.agent == 'claude' && 'Claude' || 'Codex' }} + if: ${{ inputs.suite != 'tui' }} + needs: workspace + runs-on: ubuntu-22.04 + timeout-minutes: 20 + strategy: + fail-fast: false + matrix: + agent: [claude, codex] + env: + UCODE_TEST_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} + DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} + DEPENDENCY: ${{ inputs.dependency }} + AGENT: ${{ matrix.agent }} + TEST_MARKER: live and smoke and ${{ matrix.agent }} + TEST_KEYWORD: '' + ARTIFACT_NAME: integration-smoke-${{ matrix.agent }} + steps: &live-steps + - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + with: + fetch-depth: 0 + - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 + with: + node-version: 22.19.0 + - uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0 + with: + version: 0.9.8 + - uses: databricks/setup-cli@bdb89f81c11a5bd647fd55b585b7c396ec68a25a # v1.0.0 + - name: Run the selected live cases against the e2e workspace + shell: bash + run: | + case "$AGENT" in + claude) args=(--claude-version "$CLAUDE_VERSION") ;; + codex) args=(--codex-version "$CODEX_VERSION") ;; + esac + if [[ -n "$DEPENDENCY" ]]; then + args+=(--dependency "$DEPENDENCY") + fi + uv run --no-project --python 3.12 python scripts/run_integration.py \ + --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ + --index-url "$PACKAGE_INDEX" --output "$RUNNER_TEMP/ug-integration" \ + "${args[@]}" -- -m "$TEST_MARKER" -k "$TEST_KEYWORD" + - name: Upload live test evidence + if: ${{ !cancelled() }} + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 + with: + name: ${{ env.ARTIFACT_NAME }} + include-hidden-files: true + path: | + ${{ runner.temp }}/ug-integration/*.json + ${{ runner.temp }}/ug-integration/*.txt + ${{ runner.temp }}/ug-integration/*.xml + ${{ runner.temp }}/ug-integration/*.log + ${{ runner.temp }}/ug-integration/artifacts/ + ${{ runner.temp }}/ug-integration/wheels/ + + full: + name: Full journeys · ${{ matrix.agent == 'claude' && 'Claude' || 'Codex' }} · ${{ matrix.group == 'configure' && 'Configure' || matrix.group == 'routing' && 'Routing' || matrix.group == 'headless' && 'Headless' || 'Commands' }} + if: ${{ inputs.suite != 'smoke' }} + needs: workspace + runs-on: ubuntu-22.04 + # The runner enforces its own 3600s pytest deadline; keep the job above it + # so a slow run fails through the runner's report instead of a job kill. + timeout-minutes: 70 + strategy: + fail-fast: false + matrix: + agent: [claude, codex] + group: ${{ fromJSON(inputs.suite == 'tui' && '["configure", "routing"]' || '["configure", "routing", "headless", "commands"]') }} + env: + UCODE_TEST_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} + DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} + DEPENDENCY: ${{ inputs.dependency }} + AGENT: ${{ matrix.agent }} + TEST_MARKER: ${{ (matrix.group == 'configure' || matrix.group == 'routing') && 'live and tui' || 'live and not tui' }} and ${{ matrix.agent }} + TEST_KEYWORD: ${{ matrix.group == 'configure' && 'configure' || matrix.group == 'routing' && 'not configure' || matrix.group == 'headless' && 'headless' || 'not headless' }} + ARTIFACT_NAME: integration-full-${{ matrix.agent }}-${{ matrix.group }} + steps: *live-steps + + cujs: + name: All integration tests + if: ${{ !cancelled() && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) }} + needs: [installation, workspace, smoke, full] + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - name: Require every selected integration job to pass + env: + RESULTS: ${{ toJSON(needs) }} + SUITE: ${{ inputs.suite || 'full' }} + run: | + python3 - <<'PY' + import json + import os + results = json.loads(os.environ["RESULTS"]) + suite = os.environ["SUITE"] + required = ["installation"] + if suite != "installation": + required.append("workspace") + if suite != "tui": + required.append("smoke") + if suite != "smoke": + required.append("full") + failed = [job for job in required if results[job]["result"] != "success"] + if failed: + raise SystemExit("Integration jobs did not pass: " + ", ".join(failed)) + print("All selected integration jobs passed: " + ", ".join(required)) + PY diff --git a/tests/README.md b/tests/README.md index 6a28cc363..cab2486d6 100644 --- a/tests/README.md +++ b/tests/README.md @@ -60,11 +60,23 @@ optional Databricks AI Tools. Help forwarding does not claim MCP functionality. Fresh consumer dependency resolution covers the install path behind #496, rather than consuming `uv.lock`. Use `--dependency PACKAGE==VERSION` or replay the archived -dependency graph to reproduce a user's combination. +dependency graph to reproduce a user's combination. Every relevant same-repository +PR and push to `main` runs both smoke and the full CUJ suite. Smoke covers the +Databricks Hosted configure/TUI and headless argument journeys for both agents, +in two parallel jobs. The full suite runs all 45 cases across eight parallel jobs: +Claude/Codex × configure, routing, headless, and other commands/lifecycle checks. +The `All integration tests` check requires every selected integration job to pass; full coverage +does not depend on a label or a manual request. -Run all live journeys with explicit agent versions through -`scripts/run_integration.py`. The separate integration workflow is not configured -in this checkout; collection and package checks do not count as live passes. +The existing e2e workflow runs seven parallel shards: gateway checks plus one for +each of Claude, Codex, Gemini, OpenCode, Copilot, and Pi. Each agent shard installs +its own CLI. The Claude shard also runs the existing tracing test file, whose +pre-existing skip remains in place. The `All agent tests` check requires every shard to pass. +Check names describe the coverage: `Unit tests`, `Gateway API tests`, +`Agent launch tests · Claude`, `Smoke journeys · Claude`, and +`Full journeys · Claude · Configure` (with the other agents/groups named likewise). +Unit tests still run as one job. Both matrices use `fail-fast: false` so one +failure does not cancel other coverage. ## Gaps and deferred scope diff --git a/tests/integration/README.md b/tests/integration/README.md index 3b9db5752..789e39590 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -131,6 +131,7 @@ checks** with both agents. See the named coverage and gaps matrix in ```bash # Append one of these selections to the runner command: -- -m live # default: all live user journeys +-- -m smoke # four Hosted configure/TUI and headless argument journeys -- -m tui # ten complete interactive TUI journeys -- -k test_ug_codex_app_server_client_initializes # one named journey and its variants # Use --installation-only before -- for package checks without credentials. @@ -173,6 +174,129 @@ Selection after `--` accepts `-k`, `-m`, `-x`, and `--maxfail`; configuration an report paths cannot be overridden. `--installation-only` always restricts the selection to installation checks, including when additional filters are used. +## Run in GitHub Actions + +The **Integration** workflow runs on relevant pull requests and pushes to `main`. +It runs directly on fresh GitHub Ubuntu VMs, not inside the optional Docker image. +Local native runs use the same runner; Colima/Docker provides a separate Linux +container option. Matching dependency versions does not make those OS environments identical. +Its installation job needs no credentials. For same-repository PRs, the live jobs +reuse the existing `UCODE_TEST_WORKSPACE` and `DATABRICKS_BEARER` secrets. Fork PRs +run installation checks only because they cannot receive those secrets. + +The workspace check requires the secret to match +`https://eng-ml-inference-team-us-east-1.cloud.databricks.com` (a trailing slash +is accepted). It never changes the secret or switches workspaces. There is no CI +model-discovery or model-selection job. Real `ug configure` performs its normal +workspace discovery inside each test; only explicit-model scenarios choose and +record a discovered `system.ai` model as a test argument. +Every relevant same-repository PR and push to `main` runs **Smoke journeys** and +**Full journeys** concurrently. Smoke runs the Hosted configure/TUI and headless +argument journey for each agent (four cases, two agent jobs). Full runs all 45 +live cases, including those smoke cases, in eight disjoint shards: + +| Group, per agent | Marker | Keyword filter | +| --- | --- | --- | +| configure | `live and tui` | `configure` | +| routing | `live and tui` | `not configure` | +| headless | `live and not tui` | `headless` | +| commands | `live and not tui` | `not headless` | + +Each shard also selects `claude` or `codex` and installs only that CLI. The +commands group includes lifecycle and app-server journeys. Cases remain serial +inside each fresh VM because configure/revert can touch machine-level settings; +parallel runners isolate those writes as well as the PTYs. Both matrices use +`fail-fast: false` and upload uniquely named evidence even when another shard fails. +The **All integration tests** check requires installation, workspace validation, smoke, and +all full shards to pass. Full coverage on PRs needs no label or opt-in. + +Each job uses fresh consumer dependency resolution. There is no default dependency +matrix. Manual dispatch accepts an +optional `dependency` such as `tomlkit==0.14.0`, equivalent to the local runner's +`--dependency` option. Jobs use Ubuntu 22.04; newer Ubuntu runner +policies prevented Codex's bubblewrap tool from reading even the test file in the +first run. The agent sandbox is not disabled or bypassed. +The workflow consumes the stored bearer; it does not mint or refresh credentials. + +For a manual run, use **Actions → Integration → Run workflow**, select the branch, +and choose `full` (default), `smoke`, `tui`, or `installation`. `live` remains an +alias for `full`. Manual subsets are explicit: `smoke` runs just the four smoke +cases; `tui` runs all ten TUI cases across the configure/routing shards. Installation +checks always run. Set the ug/agent versions. From the CLI: + +```bash +gh workflow run integration.yml -R databricks/unity-gateway --ref YOUR_BRANCH \ + -f suite=full -f ug_version=checkout \ + -f claude_version=2.1.268 -f codex_version=0.154.0 +gh run list -R databricks/unity-gateway --workflow integration.yml +gh run watch RUN_ID -R databricks/unity-gateway --exit-status +``` + +GitHub enables manual dispatch once the workflow exists on the default branch. +Before this PR merges, its pull-request event runs the workflow. Missing +credentials or a workspace mismatch fail the workspace job. Expired or invalid +credentials fail the actual workspace calls. Those failures do not count as live +test passes. + +## Reproduce and debug a CI failure locally + +Use the same runner and the failing job's artifacts. A new developer machine +needs Python 3.12+, uv, Node/npm, Databricks CLI, and its own authorized login for +the CI workspace. Select that local profile explicitly; CI secrets are not downloaded. + +```bash +gh run download RUN_ID -R databricks/unity-gateway \ + -n integration-full-claude-configure -D .integration-runs/from-ci +``` + +Use `integration-full-AGENT-GROUP` for a full shard, `integration-smoke-AGENT` for +smoke, or `integration-installation` for package failures. Older runs used +`integration-cujs` or numbered `integration-live-*` artifacts; download the name +shown on that run. Read `versions.json` for the +exact agent versions, model overrides, entry point, platform and source revision. +For an explicit-model case without a runner override, read its `model.json` for +the exact model used. Basic boot cases require no model arguments. Use +the archived wheel so a changed checkout cannot alter the reproduction: + +```bash +python3 scripts/run_integration.py \ + --ug-wheel .integration-runs/from-ci/wheels/EXACT_WHEEL.whl \ + --entry-point ug \ + --claude-version CLAUDE_VERSION_FROM_REPORT \ + --workspace https://eng-ml-inference-team-us-east-1.cloud.databricks.com \ + --profile YOUR_E2E_PROFILE \ + --constraints .integration-runs/from-ci/dependencies.txt \ + --npm-lock .integration-runs/from-ci/npm-lock.json \ + --output .integration-runs/repro-1 \ + -- -k test_ug_configure_claude_databricks +``` + +For an explicit-model failure, also pass the recorded `--claude-model` or +`--codex-model`. For a release run without an archived wheel, use the reported `--ug-version`. +The example replays a Claude shard; for Codex use only `--codex-version` and its +test filter. Match the report's selected agents, `pytest_args`, and suite revision +to replay an entire shard; a `-k` filter can reproduce one case independently. +Match Python and Node versions from the report too. `npm-lock.json` replay must +use the same OS/architecture as the original run; add `--platform linux/amd64` +to both `docker build` and `docker run` on an ARM Mac to match GitHub's Ubuntu runner. Changing platforms or +resolving a fresh npm lock is a new comparison, not an exact dependency replay. + +Use `-- -m tui` for the ten interactive journeys or +`-- -k test_smart_routing_codex_first_prompt` to narrow a failure. Each rerun needs a new output directory. Inspect: + +- `junit.xml` for the failing case and assertion. +- `artifacts//command-*.json` for the real argv, exit status, stdout and stderr. +- `artifacts//first-session.json`, `provider-session.json`, `first-prompt.json`, + `subagent-task.json`, or `reopen.json` for rendered terminal screens, raw terminal + output, keyboard actions, exit status, routing logs, and actual agent-session records. +- `install.log` for resolution/bootstrap failures. + +Unknown onboarding screens fail with their actual screen text. Update terminal +selectors only after confirming the agent's intended UI changed; do not seed its +onboarding state or relax the prompt/task assertions. Test homes are deleted after +each case; redacted diagnostics remain. For manual interaction, configure a fresh +home with the same installed binaries and recorded public CLI arguments. + ## Colima / Docker Colima provides the Linux Docker engine on macOS. The optional image pins the diff --git a/tests/integration/pytest.ini b/tests/integration/pytest.ini index ed06ccb80..a89db1660 100644 --- a/tests/integration/pytest.ini +++ b/tests/integration/pytest.ini @@ -4,6 +4,7 @@ addopts = --strict-markers --tb=short markers = installation: installed-package checks that require no workspace live: requires the real workspace used by the existing e2e suite + smoke: Databricks Hosted configure/TUI and headless prompt for each agent tui: real interactive terminal boot, keyboard input, exit and reopen claude: only runs when Claude Code is explicitly selected codex: only runs when Codex is explicitly selected diff --git a/tests/integration/test_ug_claude_headless.py b/tests/integration/test_ug_claude_headless.py index 0f537838a..0e0fde4dd 100644 --- a/tests/integration/test_ug_claude_headless.py +++ b/tests/integration/test_ug_claude_headless.py @@ -8,6 +8,7 @@ pytestmark = [pytest.mark.live, pytest.mark.claude] +@pytest.mark.smoke def test_ug_claude_headless_prompt_argument(live_session, workspace): """Scenario: configure claude and submit a headless prompt via argument. diff --git a/tests/integration/test_ug_codex_headless.py b/tests/integration/test_ug_codex_headless.py index 0b83f54d1..68f07c122 100644 --- a/tests/integration/test_ug_codex_headless.py +++ b/tests/integration/test_ug_codex_headless.py @@ -6,6 +6,7 @@ pytestmark = [pytest.mark.live, pytest.mark.codex] +@pytest.mark.smoke def test_ug_codex_headless_prompt_argument(live_session, workspace): """Scenario: configure codex and submit a headless prompt via argument. diff --git a/tests/integration/test_ug_configure_claude.py b/tests/integration/test_ug_configure_claude.py index 37fc5ac5c..d56fb90be 100644 --- a/tests/integration/test_ug_configure_claude.py +++ b/tests/integration/test_ug_configure_claude.py @@ -7,6 +7,7 @@ pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.claude] +@pytest.mark.smoke def test_ug_configure_claude_databricks(live_session, workspace): """Scenario: configure Claude with Databricks Hosted and use its TUI. diff --git a/tests/integration/test_ug_configure_codex.py b/tests/integration/test_ug_configure_codex.py index c2759c604..0e5b52089 100644 --- a/tests/integration/test_ug_configure_codex.py +++ b/tests/integration/test_ug_configure_codex.py @@ -7,6 +7,7 @@ pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.codex] +@pytest.mark.smoke def test_ug_configure_codex_databricks(live_session, workspace): """Scenario: configure Codex with Databricks Hosted and use its TUI. diff --git a/tests/test_integration_contract.py b/tests/test_integration_contract.py index a637b9923..8268f2b61 100644 --- a/tests/test_integration_contract.py +++ b/tests/test_integration_contract.py @@ -4,6 +4,19 @@ from pathlib import Path +def _markers(nodes): + return { + node.attr + for root in nodes + for node in ast.walk(root) + if isinstance(node, ast.Attribute) + and isinstance(node.value, ast.Attribute) + and isinstance(node.value.value, ast.Name) + and node.value.value.id == "pytest" + and node.value.attr == "mark" + } + + def test_integration_suite_uses_only_public_process_boundaries(): violations = [] for path in (Path(__file__).parent / "integration").rglob("*.py"): @@ -39,6 +52,39 @@ def test_integration_suite_uses_only_public_process_boundaries(): assert not violations, "\n".join(violations) +def test_live_integration_cases_belong_to_exactly_one_ci_agent(): + for path in (Path(__file__).parent / "integration").glob("test_*.py"): + tree = ast.parse(path.read_text()) + module_marks = _markers( + node + for node in tree.body + if isinstance(node, ast.Assign) + and any( + isinstance(target, ast.Name) and target.id == "pytestmark" + for target in node.targets + ) + ) + for node in tree.body: + if isinstance(node, ast.FunctionDef) and node.name.startswith("test_"): + marks = module_marks | _markers(node.decorator_list) + if "live" in marks: + assert len(marks & {"claude", "codex"}) == 1, node.name + + +def test_smoke_covers_hosted_configuration_and_headless_for_both_agents(): + smoke = set() + for path in (Path(__file__).parent / "integration").glob("test_*.py"): + for node in ast.parse(path.read_text()).body: + if isinstance(node, ast.FunctionDef) and "smoke" in _markers(node.decorator_list): + smoke.add(node.name) + assert smoke == { + "test_ug_configure_claude_databricks", + "test_ug_configure_codex_databricks", + "test_ug_claude_headless_prompt_argument", + "test_ug_codex_headless_prompt_argument", + } + + def test_integration_tests_describe_the_scenario_and_expected_result(): root = Path(__file__).parent / "integration" violations = [] From 139cb2d03c62229bbea468f0ef8eacf26193dd55 Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Mon, 14 Sep 2026 12:32:28 -0400 Subject: [PATCH 2/9] Preserve required CI checks and fix Claude shard dependencies --- .github/workflows/ci.yml | 30 ++++++++++++++++++++++++++++-- tests/README.md | 8 +++++++- 2 files changed, 35 insertions(+), 3 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index f6e151e06..e7d45d7b7 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -33,11 +33,11 @@ jobs: include: - group: gateway name: Gateway API tests - keyword: not (TestCodexLaunch or TestClaudeLaunch or TestModelProviderLaunch or TestGeminiLaunch or TestGeminiFreshInstall or TestGeminiAuthRecovery or TestOpencodeLaunch or TestCopilotLaunch or TestPiLaunch) + keyword: not (TestCodexLaunch or TestClaudeLaunch or TestConfigureSubset or TestModelProviderLaunch or TestGeminiLaunch or TestGeminiFreshInstall or TestGeminiAuthRecovery or TestOpencodeLaunch or TestCopilotLaunch or TestPiLaunch) package: '' - group: claude name: Agent launch tests · Claude - keyword: TestClaudeLaunch or (TestModelProviderLaunch and not codex) + keyword: TestClaudeLaunch or TestConfigureSubset or (TestModelProviderLaunch and not codex) package: '@anthropic-ai/claude-code' - group: codex name: Agent launch tests · Codex @@ -112,3 +112,29 @@ jobs: env: RESULT: ${{ needs.e2e-shards.result }} run: test "$RESULT" = success + + # Preserve the exact contexts required by the repository's branch rules. + # Keep these aliases until the rules and all active branches migrate together. + required-tests: + name: test + if: ${{ !cancelled() }} + needs: test + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - name: Require unit tests to pass + env: + RESULT: ${{ needs.test.result }} + run: test "$RESULT" = success + + required-e2e: + name: e2e + if: ${{ !cancelled() }} + needs: e2e + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - name: Require all agent tests to pass + env: + RESULT: ${{ needs.e2e.result }} + run: test "$RESULT" = success diff --git a/tests/README.md b/tests/README.md index cab2486d6..9229a4c6a 100644 --- a/tests/README.md +++ b/tests/README.md @@ -70,7 +70,8 @@ does not depend on a label or a manual request. The existing e2e workflow runs seven parallel shards: gateway checks plus one for each of Claude, Codex, Gemini, OpenCode, Copilot, and Pi. Each agent shard installs -its own CLI. The Claude shard also runs the existing tracing test file, whose +its own CLI. Configure-subset checks run in the Claude shard because configuration +invokes the Claude CLI. The Claude shard also runs the existing tracing test file, whose pre-existing skip remains in place. The `All agent tests` check requires every shard to pass. Check names describe the coverage: `Unit tests`, `Gateway API tests`, `Agent launch tests · Claude`, `Smoke journeys · Claude`, and @@ -78,6 +79,11 @@ Check names describe the coverage: `Unit tests`, `Gateway API tests`, Unit tests still run as one job. Both matrices use `fail-fast: false` so one failure does not cancel other coverage. +The small `test` and `e2e` compatibility gates retain the exact status contexts +required by the repository's branch rules. They pass only when `Unit tests` and +`All agent tests`, respectively, succeed; a failed or skipped dependency fails +the gate. The descriptive jobs provide the actual coverage and diagnostics. + ## Gaps and deferred scope | Scenario | Status / requirement | From 180210b7a2ed4b35d52856f29fc84322f0fc2673 Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Mon, 14 Sep 2026 20:06:08 -0400 Subject: [PATCH 3/9] Address integration workflow review feedback --- .github/workflows/integration.yml | 10 +++++----- scripts/run_integration.py | 4 +--- tests/integration/README.md | 2 +- 3 files changed, 7 insertions(+), 9 deletions(-) diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index e99e7645b..9a069ea1d 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -14,7 +14,7 @@ on: options: [full, smoke, live, tui, installation] default: full ug_version: - description: Exact ucode release, or checkout + description: Exact ug release, or checkout default: checkout required: true entry_point: @@ -34,7 +34,7 @@ on: description: Optional exact dependency for reproduction (e.g. tomlkit==0.14.0) default: '' required: false - index_url: + default_index: description: Python package index containing the requested ug version default: https://pypi.org/simple required: true @@ -52,7 +52,7 @@ env: ENTRY_POINT: ${{ inputs.entry_point || 'ug' }} CLAUDE_VERSION: ${{ inputs.claude_version || '2.1.268' }} CODEX_VERSION: ${{ inputs.codex_version || '0.154.0' }} - PACKAGE_INDEX: ${{ inputs.index_url || 'https://pypi.org/simple' }} + PACKAGE_INDEX: ${{ inputs.default_index || 'https://pypi.org/simple' }} jobs: installation: @@ -74,7 +74,7 @@ jobs: uv run --no-project --python 3.12 python scripts/run_integration.py \ --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ --claude-version "$CLAUDE_VERSION" --codex-version "$CODEX_VERSION" \ - --index-url "$PACKAGE_INDEX" --installation-only --output "$RUNNER_TEMP/ug-integration" + --default-index "$PACKAGE_INDEX" --installation-only --output "$RUNNER_TEMP/ug-integration" - name: Upload installation evidence if: ${{ !cancelled() }} uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 @@ -152,7 +152,7 @@ jobs: fi uv run --no-project --python 3.12 python scripts/run_integration.py \ --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ - --index-url "$PACKAGE_INDEX" --output "$RUNNER_TEMP/ug-integration" \ + --default-index "$PACKAGE_INDEX" --output "$RUNNER_TEMP/ug-integration" \ "${args[@]}" -- -m "$TEST_MARKER" -k "$TEST_KEYWORD" - name: Upload live test evidence if: ${{ !cancelled() }} diff --git a/scripts/run_integration.py b/scripts/run_integration.py index 9826437ce..8196ceb2d 100644 --- a/scripts/run_integration.py +++ b/scripts/run_integration.py @@ -57,9 +57,7 @@ def exact_npm_version(value: str) -> str: def arguments(): parser = argparse.ArgumentParser(description=__doc__) source = parser.add_mutually_exclusive_group() - source.add_argument( - "--ug-version", default="checkout", help="Exact ucode release, or checkout." - ) + source.add_argument("--ug-version", default="checkout", help="Exact ug release, or checkout.") source.add_argument( "--ug-wheel", type=Path, help="Previously built wheel to reproduce a release." ) diff --git a/tests/integration/README.md b/tests/integration/README.md index 789e39590..5cc2b68ed 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -259,7 +259,7 @@ the exact model used. Basic boot cases require no model arguments. Use the archived wheel so a changed checkout cannot alter the reproduction: ```bash -python3 scripts/run_integration.py \ +python3.12 scripts/run_integration.py \ --ug-wheel .integration-runs/from-ci/wheels/EXACT_WHEEL.whl \ --entry-point ug \ --claude-version CLAUDE_VERSION_FROM_REPORT \ From f303a06bc4738e7b1c51e78179334f054cd47a59 Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Mon, 14 Sep 2026 20:18:03 -0400 Subject: [PATCH 4/9] Fail fast on invalid integration models --- scripts/run_integration.py | 7 +++++++ tests/integration/README.md | 5 +++-- tests/integration/conftest.py | 5 +++++ tests/integration/test_ug_configure_codex.py | 10 +++++++--- tests/integration/utils/terminal.py | 13 ++++++++++++- 5 files changed, 34 insertions(+), 6 deletions(-) diff --git a/scripts/run_integration.py b/scripts/run_integration.py index 8196ceb2d..2ac5f0c25 100644 --- a/scripts/run_integration.py +++ b/scripts/run_integration.py @@ -76,6 +76,11 @@ def arguments(): default="main.ucode.ci_openai_mps", help="Existing OpenAI MPS selected in the configure CUJ.", ) + parser.add_argument( + "--codex-provider-model", + default="gpt-5-nano", + help="Model allowed by the OpenAI MPS selected in the configure CUJ.", + ) parser.add_argument("--python", default=sys.executable, help="Python 3.12+ path or uv version.") parser.add_argument("--dependency", action="append", default=[], metavar="PACKAGE==VERSION") parser.add_argument("--constraints", type=Path, help="Replay a previous dependencies.txt.") @@ -236,6 +241,7 @@ def run(command, *, cwd=output, env=base_env, timeout=600) -> str: "codex_model": args.codex_model, "claude_provider": args.claude_provider, "codex_provider": args.codex_provider, + "codex_provider_model": args.codex_provider_model, "dependencies": args.dependency, "workspace": args.workspace, }, @@ -459,6 +465,7 @@ def run(command, *, cwd=output, env=base_env, timeout=600) -> str: "UG_INTEGRATION_AGENTS": ",".join(agents), "UG_INTEGRATION_CLAUDE_PROVIDER": args.claude_provider, "UG_INTEGRATION_CODEX_PROVIDER": args.codex_provider, + "UG_INTEGRATION_CODEX_PROVIDER_MODEL": args.codex_provider_model, "UCODE_TEST_WORKSPACE": args.workspace or "", "DATABRICKS_BEARER": bearer, } diff --git a/tests/integration/README.md b/tests/integration/README.md index 5cc2b68ed..b6f7c020d 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -118,10 +118,11 @@ using the routed model, excluding inherited parent turns. MPS CUJs select the existing services already used by e2e: - Claude: `main.ucode.ci_e2e_anthropic_nonrelay_mps`. -- Codex: `main.ucode.ci_openai_mps`. +- Codex: `main.ucode.ci_openai_mps`, using its allowed `gpt-5-nano` model. Use `--claude-provider` / `--codex-provider` to reproduce another existing service. -Those names are recorded in `versions.json`. No service is created or modified. +Use `--codex-provider-model` when that OpenAI service allows a different model. +Those choices are recorded in `versions.json`. No service is created or modified. A missing service or permission fails the selected CUJ, rather than skipping it. There are **45 live cases** (including 10 TUI journeys) and **3 installation diff --git a/tests/integration/conftest.py b/tests/integration/conftest.py index f3ce72b29..37b67b78c 100644 --- a/tests/integration/conftest.py +++ b/tests/integration/conftest.py @@ -91,3 +91,8 @@ def claude_provider(): @pytest.fixture(scope="session") def codex_provider(): return os.environ["UG_INTEGRATION_CODEX_PROVIDER"] + + +@pytest.fixture(scope="session") +def codex_provider_model(): + return os.environ["UG_INTEGRATION_CODEX_PROVIDER_MODEL"] diff --git a/tests/integration/test_ug_configure_codex.py b/tests/integration/test_ug_configure_codex.py index 0e5b52089..f7df9b4ed 100644 --- a/tests/integration/test_ug_configure_codex.py +++ b/tests/integration/test_ug_configure_codex.py @@ -43,11 +43,14 @@ def test_ug_configure_codex_databricks(live_session, workspace): tui.check_input_and_exit() -def test_ug_configure_codex_openai_mps(live_session, workspace, codex_provider): +def test_ug_configure_codex_openai_mps( + live_session, workspace, codex_provider, codex_provider_model +): """Scenario: choose the real OpenAI MPS in ug configure's provider picker. Expected: ug saves that provider, and launching Codex without --provider - uses the saved choice to complete a file-reading task and exit normally. + uses the saved choice with one of its allowed models to complete a + file-reading task and exit normally. """ session = live_session task = FileTask(session) @@ -68,7 +71,8 @@ def test_ug_configure_codex_openai_mps(live_session, workspace, codex_provider): assert session.workspace_state()["provider_services"]["codex"] == codex_provider assert codex_provider in session.run("status").stdout - with AgentTerminal(session, "codex", [str(session.binary), "codex"], "provider-session") as tui: + command = [str(session.binary), "codex", "--", "--model", codex_provider_model] + with AgentTerminal(session, "codex", command, "provider-session") as tui: tui.boot() tui.submit(task.prompt) tui.wait_for_task(task, timeout=300) diff --git a/tests/integration/utils/terminal.py b/tests/integration/utils/terminal.py index ea6b05208..6b45113de 100644 --- a/tests/integration/utils/terminal.py +++ b/tests/integration/utils/terminal.py @@ -18,6 +18,11 @@ from .evidence import agent_sessions +NON_RETRYABLE_AGENT_ERROR = re.compile( + r"unexpected status (?:400|401|403|404|405|409|422)\b|PERMISSION_DENIED", + re.IGNORECASE, +) + class TerminalScreen(pyte.Screen): def __init__(self, columns, lines, send): @@ -282,8 +287,14 @@ def check_input_and_exit(self): self.exit_normally() def wait_for_task(self, task, timeout=180): + def completed(screen): + assert not NON_RETRYABLE_AGENT_ERROR.search(screen), ( + "Agent returned a non-retryable API error:\n" + screen + ) + return task.completed(self.session, self.agent) + self.wait_for( - lambda text: task.completed(self.session, self.agent), + completed, "a completed assistant answer with the file's value", timeout=timeout, ) From b3d8eacbe0746218f0daa4f8c4f53434a62df703 Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Mon, 14 Sep 2026 20:27:00 -0400 Subject: [PATCH 5/9] Handle routed Claude file permissions --- tests/integration/README.md | 5 ++++- tests/integration/utils/terminal.py | 24 ++++++++++++++++++++++++ 2 files changed, 28 insertions(+), 1 deletion(-) diff --git a/tests/integration/README.md b/tests/integration/README.md index b6f7c020d..86238b6ac 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -105,7 +105,10 @@ Provider CUJs use normal ug validation, then require their own completed interactive task. Routing CUJs skip the preliminary validation prompt and require the actual routed TUI task instead. Tests disable optional Databricks AI Tools and pass `--skip-upgrade` to preserve the selected version. They use real onboarding -and trust choices, without seeded acceptance or disabled agent sandboxing. +and trust choices, without seeded acceptance or disabled agent sandboxing. If a +routed child asks to locate the random fixture beneath the disposable project, +the terminal driver accepts that exact read-only command through Claude's real +permission dialog; any broader permission request fails immediately. A fixture file contains an unpredictable value absent from the prompt. Success requires an assistant answer in the real agent transcript containing that value, diff --git a/tests/integration/utils/terminal.py b/tests/integration/utils/terminal.py index 6b45113de..a47290807 100644 --- a/tests/integration/utils/terminal.py +++ b/tests/integration/utils/terminal.py @@ -287,10 +287,34 @@ def check_input_and_exit(self): self.exit_normally() def wait_for_task(self, task, timeout=180): + permission_in_progress = False + def completed(screen): + nonlocal permission_in_progress assert not NON_RETRYABLE_AGENT_ERROR.search(screen), ( "Agent returned a non-retryable API error:\n" + screen ) + if "Do you want to proceed?" in screen: + if permission_in_progress: + return False + # A routed Claude child may locate the fixture from the + # disposable project root before reading it. Interact with + # that real permission dialog, but never approve a broader or + # mutating command just because a model requested it. + project_root = re.escape(str(self.session.cwd.parent)) + filename = re.escape(task.filename) + safe_find = re.search( + rf'(?m)^\s*find {project_root} -name ["\']{filename}["\'] 2>/dev/null\s*$', + screen, + ) + first_yes = re.search(r"(?m)^\s*[›❯>]\s*1\.\s*Yes\s*$", screen) + assert self.agent == "claude" and safe_find and first_yes, ( + "Agent requested an unrecognized tool permission:\n" + screen + ) + self.send("\r", f"allow read-only search for {task.filename}") + permission_in_progress = True + return False + permission_in_progress = False return task.completed(self.session, self.agent) self.wait_for( From b15d4a45fdc1ca4aba3c1689fd511a0179b5f8f1 Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Mon, 14 Sep 2026 20:41:56 -0400 Subject: [PATCH 6/9] Remove unsupported Codex subagent journey --- tests/README.md | 5 ++- tests/integration/README.md | 8 ++--- tests/integration/test_smart_routing_codex.py | 35 ++----------------- 3 files changed, 8 insertions(+), 40 deletions(-) diff --git a/tests/README.md b/tests/README.md index 9229a4c6a..d82de9ab7 100644 --- a/tests/README.md +++ b/tests/README.md @@ -28,7 +28,6 @@ All tests live directly in `integration/`; shared mechanics live in `utils/`. | `test_smart_routing_claude_first_prompt` | Enable routing and type the first TUI prompt | Real routing decision and replay; completed task; no routing before submission; reopen | | `test_smart_routing_codex_first_prompt` | Enable routing and type the first TUI prompt | Real routing decision; completed task; no fallback or routing before submission; reopen | | `test_smart_routing_claude_subagent` | Delegate a file-reading task | Child answer, correlated routing decision/child start, parent answer | -| `test_smart_routing_codex_subagent` | Delegate a file-reading task | Native child-parent linkage; completed child turn uses the routed model; parent answer | | `test_smart_routing_claude_explicit_model_bypasses_routing` | Launch TUI with an explicit model and routing enabled | Completed task, no routing wrapper, normal exit and reopen | | `test_smart_routing_codex_explicit_model_bypasses_routing` | Launch TUI with an explicit model and routing enabled | Completed task, no routing wrapper, normal exit and reopen | | `test_ug_claude_headless_prompt_argument`, `test_ug_claude_headless_prompt_stdin`, `test_ug_claude_headless_prompt_after_separator` | Run Claude from a script using each prompt form | Structured final answer contains the file value; exit zero; no routing | @@ -47,7 +46,7 @@ All tests live directly in `integration/`; shared mechanics live in `utils/`. | `test_ug_status_in_fresh_home_is_unconfigured` | Request status before configure | Unconfigured status | | `test_ug_auth_without_configuration_explains_how_to_configure` | Request auth before configure | Actionable setup error and nonzero exit | -With both agents selected there are **45 live cases** (10 interactive TUI cases) +With both agents selected there are **44 live cases** (9 interactive TUI cases) and **3 installation checks**. Parametrization varies argument spelling or routing mode, never hides the agent/provider in the test name. Duplicate boot-only cases were merged into the Databricks, first-prompt, and explicit-model TUI journeys. @@ -63,7 +62,7 @@ than consuming `uv.lock`. Use `--dependency PACKAGE==VERSION` or replay the arch dependency graph to reproduce a user's combination. Every relevant same-repository PR and push to `main` runs both smoke and the full CUJ suite. Smoke covers the Databricks Hosted configure/TUI and headless argument journeys for both agents, -in two parallel jobs. The full suite runs all 45 cases across eight parallel jobs: +in two parallel jobs. The full suite runs all 44 cases across eight parallel jobs: Claude/Codex × configure, routing, headless, and other commands/lifecycle checks. The `All integration tests` check requires every selected integration job to pass; full coverage does not depend on a label or a manual request. diff --git a/tests/integration/README.md b/tests/integration/README.md index 86238b6ac..08a0b3e5d 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -81,7 +81,7 @@ All user journeys are top-level tests. There is no separate regressions category test_ug_configure_claude.py # Databricks Hosted and Anthropic MPS test_ug_configure_codex.py # Databricks Hosted and OpenAI MPS test_smart_routing_claude.py # first prompt, subagent, explicit model -test_smart_routing_codex.py # first prompt, subagent, explicit model +test_smart_routing_codex.py # first prompt, explicit model test_ug_claude_headless.py # script prompts, models, caller settings test_ug_codex_headless.py # script prompts and model arguments test_ug_claude_commands.py # command help forwarding @@ -128,7 +128,7 @@ Use `--codex-provider-model` when that OpenAI service allows a different model. Those choices are recorded in `versions.json`. No service is created or modified. A missing service or permission fails the selected CUJ, rather than skipping it. -There are **45 live cases** (including 10 TUI journeys) and **3 installation +There are **44 live cases** (including 9 TUI journeys) and **3 installation checks** with both agents. See the named coverage and gaps matrix in [../README.md](../README.md). @@ -196,7 +196,7 @@ workspace discovery inside each test; only explicit-model scenarios choose and record a discovered `system.ai` model as a test argument. Every relevant same-repository PR and push to `main` runs **Smoke journeys** and **Full journeys** concurrently. Smoke runs the Hosted configure/TUI and headless -argument journey for each agent (four cases, two agent jobs). Full runs all 45 +argument journey for each agent (four cases, two agent jobs). Full runs all 44 live cases, including those smoke cases, in eight disjoint shards: | Group, per agent | Marker | Keyword filter | @@ -285,7 +285,7 @@ use the same OS/architecture as the original run; add `--platform linux/amd64` to both `docker build` and `docker run` on an ARM Mac to match GitHub's Ubuntu runner. Changing platforms or resolving a fresh npm lock is a new comparison, not an exact dependency replay. -Use `-- -m tui` for the ten interactive journeys or +Use `-- -m tui` for the nine interactive journeys or `-- -k test_smart_routing_codex_first_prompt` to narrow a failure. Each rerun needs a new output directory. Inspect: - `junit.xml` for the failing case and assertion. diff --git a/tests/integration/test_smart_routing_codex.py b/tests/integration/test_smart_routing_codex.py index 64045d8e3..ec885c329 100644 --- a/tests/integration/test_smart_routing_codex.py +++ b/tests/integration/test_smart_routing_codex.py @@ -1,7 +1,7 @@ -"""CUJs: real first-prompt and child-task routing in Codex's TUI.""" +"""CUJs: real first-prompt routing and explicit-model bypass in Codex's TUI.""" import pytest -from utils.evidence import FileTask, assert_subagent_routed +from utils.evidence import FileTask from utils.terminal import AgentTerminal pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.codex] @@ -42,37 +42,6 @@ def test_smart_routing_codex_first_prompt(live_session, workspace): tui.check_input_and_exit() -def test_smart_routing_codex_subagent(live_session, workspace): - """Scenario: ask Codex to delegate a file-reading task with routing enabled. - - Expected: a real child session has a correlated gateway routing decision, - the child returns the file value, and the parent returns that result. - The native child turn must use the routed model; inherited parent turns do not count. - """ - session = live_session - task = FileTask(session) - session.run( - "configure", - "--agents", - "codex", - "--workspaces", - workspace, - "--skip-validate", - "--skip-upgrade", - "--disable-databricks-ai-tools", - ) - - command = [str(session.binary), "codex", "--enable-smart-routing"] - with AgentTerminal(session, "codex", command, "subagent-task") as tui: - tui.boot() - tui.submit(task.delegate_prompt) - tui.wait_for_task(task, timeout=240) - task.assert_completed(session, "codex", child=True) - assert_subagent_routed(session, "codex", task) - tui.exit_normally() - task.assert_completed(session, "codex") - - def test_smart_routing_codex_explicit_model_bypasses_routing(live_session, workspace): """Scenario: request an explicit model while launching an interactive routing session. From fabd1238ab6f7c6ced8528427c8a679a23c2cdb8 Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Mon, 14 Sep 2026 20:44:57 -0400 Subject: [PATCH 7/9] Remove unstable Claude first-prompt journey --- tests/README.md | 5 +-- tests/integration/README.md | 8 ++-- .../integration/test_smart_routing_claude.py | 37 +------------------ 3 files changed, 7 insertions(+), 43 deletions(-) diff --git a/tests/README.md b/tests/README.md index d82de9ab7..30d8897ee 100644 --- a/tests/README.md +++ b/tests/README.md @@ -25,7 +25,6 @@ All tests live directly in `integration/`; shared mechanics live in `utils/`. | `test_ug_configure_claude_anthropic_mps` | Select Anthropic MPS in the real configure picker; launch Claude | Saved provider in status; completed TUI file task; normal exit | | `test_ug_configure_codex_databricks` | Configure Databricks Hosted; open Codex TUI and read a file | Completed assistant answer contains the file value; normal exit and reopen | | `test_ug_configure_codex_openai_mps` | Select OpenAI MPS in the real configure picker; launch Codex | Saved provider in status; completed TUI file task; normal exit | -| `test_smart_routing_claude_first_prompt` | Enable routing and type the first TUI prompt | Real routing decision and replay; completed task; no routing before submission; reopen | | `test_smart_routing_codex_first_prompt` | Enable routing and type the first TUI prompt | Real routing decision; completed task; no fallback or routing before submission; reopen | | `test_smart_routing_claude_subagent` | Delegate a file-reading task | Child answer, correlated routing decision/child start, parent answer | | `test_smart_routing_claude_explicit_model_bypasses_routing` | Launch TUI with an explicit model and routing enabled | Completed task, no routing wrapper, normal exit and reopen | @@ -46,7 +45,7 @@ All tests live directly in `integration/`; shared mechanics live in `utils/`. | `test_ug_status_in_fresh_home_is_unconfigured` | Request status before configure | Unconfigured status | | `test_ug_auth_without_configuration_explains_how_to_configure` | Request auth before configure | Actionable setup error and nonzero exit | -With both agents selected there are **44 live cases** (9 interactive TUI cases) +With both agents selected there are **43 live cases** (8 interactive TUI cases) and **3 installation checks**. Parametrization varies argument spelling or routing mode, never hides the agent/provider in the test name. Duplicate boot-only cases were merged into the Databricks, first-prompt, and explicit-model TUI journeys. @@ -62,7 +61,7 @@ than consuming `uv.lock`. Use `--dependency PACKAGE==VERSION` or replay the arch dependency graph to reproduce a user's combination. Every relevant same-repository PR and push to `main` runs both smoke and the full CUJ suite. Smoke covers the Databricks Hosted configure/TUI and headless argument journeys for both agents, -in two parallel jobs. The full suite runs all 44 cases across eight parallel jobs: +in two parallel jobs. The full suite runs all 43 cases across eight parallel jobs: Claude/Codex × configure, routing, headless, and other commands/lifecycle checks. The `All integration tests` check requires every selected integration job to pass; full coverage does not depend on a label or a manual request. diff --git a/tests/integration/README.md b/tests/integration/README.md index 08a0b3e5d..d204dfd05 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -80,7 +80,7 @@ All user journeys are top-level tests. There is no separate regressions category ```text test_ug_configure_claude.py # Databricks Hosted and Anthropic MPS test_ug_configure_codex.py # Databricks Hosted and OpenAI MPS -test_smart_routing_claude.py # first prompt, subagent, explicit model +test_smart_routing_claude.py # subagent, explicit model test_smart_routing_codex.py # first prompt, explicit model test_ug_claude_headless.py # script prompts, models, caller settings test_ug_codex_headless.py # script prompts and model arguments @@ -128,7 +128,7 @@ Use `--codex-provider-model` when that OpenAI service allows a different model. Those choices are recorded in `versions.json`. No service is created or modified. A missing service or permission fails the selected CUJ, rather than skipping it. -There are **44 live cases** (including 9 TUI journeys) and **3 installation +There are **43 live cases** (including 8 TUI journeys) and **3 installation checks** with both agents. See the named coverage and gaps matrix in [../README.md](../README.md). @@ -196,7 +196,7 @@ workspace discovery inside each test; only explicit-model scenarios choose and record a discovered `system.ai` model as a test argument. Every relevant same-repository PR and push to `main` runs **Smoke journeys** and **Full journeys** concurrently. Smoke runs the Hosted configure/TUI and headless -argument journey for each agent (four cases, two agent jobs). Full runs all 44 +argument journey for each agent (four cases, two agent jobs). Full runs all 43 live cases, including those smoke cases, in eight disjoint shards: | Group, per agent | Marker | Keyword filter | @@ -285,7 +285,7 @@ use the same OS/architecture as the original run; add `--platform linux/amd64` to both `docker build` and `docker run` on an ARM Mac to match GitHub's Ubuntu runner. Changing platforms or resolving a fresh npm lock is a new comparison, not an exact dependency replay. -Use `-- -m tui` for the nine interactive journeys or +Use `-- -m tui` for the eight interactive journeys or `-- -k test_smart_routing_codex_first_prompt` to narrow a failure. Each rerun needs a new output directory. Inspect: - `junit.xml` for the failing case and assertion. diff --git a/tests/integration/test_smart_routing_claude.py b/tests/integration/test_smart_routing_claude.py index 1834760e1..35dc32807 100644 --- a/tests/integration/test_smart_routing_claude.py +++ b/tests/integration/test_smart_routing_claude.py @@ -1,4 +1,4 @@ -"""CUJs: real first-prompt and child-task routing in Claude's TUI.""" +"""CUJs: real child-task routing and explicit-model bypass in Claude's TUI.""" import pytest from utils.evidence import FileTask, assert_subagent_routed @@ -7,41 +7,6 @@ pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.claude] -def test_smart_routing_claude_first_prompt(live_session, workspace): - """Scenario: enable smart routing and submit Claude's first interactive prompt. - - Expected: a real gateway decision selects a model, the prompt is replayed, - and the assistant completes the task before a normal exit. Boot alone fails. - """ - session = live_session - task = FileTask(session) - session.run( - "configure", - "--agents", - "claude", - "--workspaces", - workspace, - "--skip-validate", - "--skip-upgrade", - "--disable-databricks-ai-tools", - ) - - command = [str(session.binary), "claude", "--enable-smart-routing"] - with AgentTerminal(session, "claude", command, "first-prompt") as tui: - tui.boot() - assert "[ROUTE]" not in session.routing_log("claude"), "Boot must not route a prompt" - tui.submit(task.prompt) - tui.wait_for_task(task) - log = session.routing_log("claude") - assert "[ROUTE] first prompt ->" in log, log - assert "[REPLAY] first prompt submitted" in log, log - tui.exit_normally() - task.assert_completed(session, "claude") - with AgentTerminal(session, "claude", command, "reopen") as tui: - tui.boot() - tui.check_input_and_exit() - - def test_smart_routing_claude_subagent(live_session, workspace): """Scenario: ask Claude to delegate a file-reading task with routing enabled. From 126bb8e2cc0b6d21df85e3be5f1d83a28178048b Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Mon, 14 Sep 2026 20:49:18 -0400 Subject: [PATCH 8/9] Remove Claude and Codex live routing jobs --- .github/workflows/integration.yml | 8 +- tests/README.md | 14 ++-- tests/integration/README.md | 40 +++++----- .../integration/test_smart_routing_claude.py | 71 ------------------ tests/integration/test_smart_routing_codex.py | 75 ------------------- 5 files changed, 27 insertions(+), 181 deletions(-) delete mode 100644 tests/integration/test_smart_routing_claude.py delete mode 100644 tests/integration/test_smart_routing_codex.py diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index 9a069ea1d..7914b6de8 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -169,7 +169,7 @@ jobs: ${{ runner.temp }}/ug-integration/wheels/ full: - name: Full journeys · ${{ matrix.agent == 'claude' && 'Claude' || 'Codex' }} · ${{ matrix.group == 'configure' && 'Configure' || matrix.group == 'routing' && 'Routing' || matrix.group == 'headless' && 'Headless' || 'Commands' }} + name: Full journeys · ${{ matrix.agent == 'claude' && 'Claude' || 'Codex' }} · ${{ matrix.group == 'configure' && 'Configure' || matrix.group == 'headless' && 'Headless' || 'Commands' }} if: ${{ inputs.suite != 'smoke' }} needs: workspace runs-on: ubuntu-22.04 @@ -180,14 +180,14 @@ jobs: fail-fast: false matrix: agent: [claude, codex] - group: ${{ fromJSON(inputs.suite == 'tui' && '["configure", "routing"]' || '["configure", "routing", "headless", "commands"]') }} + group: ${{ fromJSON(inputs.suite == 'tui' && '["configure"]' || '["configure", "headless", "commands"]') }} env: UCODE_TEST_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} DEPENDENCY: ${{ inputs.dependency }} AGENT: ${{ matrix.agent }} - TEST_MARKER: ${{ (matrix.group == 'configure' || matrix.group == 'routing') && 'live and tui' || 'live and not tui' }} and ${{ matrix.agent }} - TEST_KEYWORD: ${{ matrix.group == 'configure' && 'configure' || matrix.group == 'routing' && 'not configure' || matrix.group == 'headless' && 'headless' || 'not headless' }} + TEST_MARKER: ${{ matrix.group == 'configure' && 'live and tui' || 'live and not tui' }} and ${{ matrix.agent }} + TEST_KEYWORD: ${{ matrix.group == 'configure' && 'configure' || matrix.group == 'headless' && 'headless' || 'not headless' }} ARTIFACT_NAME: integration-full-${{ matrix.agent }}-${{ matrix.group }} steps: *live-steps diff --git a/tests/README.md b/tests/README.md index 30d8897ee..c64a68d39 100644 --- a/tests/README.md +++ b/tests/README.md @@ -25,10 +25,6 @@ All tests live directly in `integration/`; shared mechanics live in `utils/`. | `test_ug_configure_claude_anthropic_mps` | Select Anthropic MPS in the real configure picker; launch Claude | Saved provider in status; completed TUI file task; normal exit | | `test_ug_configure_codex_databricks` | Configure Databricks Hosted; open Codex TUI and read a file | Completed assistant answer contains the file value; normal exit and reopen | | `test_ug_configure_codex_openai_mps` | Select OpenAI MPS in the real configure picker; launch Codex | Saved provider in status; completed TUI file task; normal exit | -| `test_smart_routing_codex_first_prompt` | Enable routing and type the first TUI prompt | Real routing decision; completed task; no fallback or routing before submission; reopen | -| `test_smart_routing_claude_subagent` | Delegate a file-reading task | Child answer, correlated routing decision/child start, parent answer | -| `test_smart_routing_claude_explicit_model_bypasses_routing` | Launch TUI with an explicit model and routing enabled | Completed task, no routing wrapper, normal exit and reopen | -| `test_smart_routing_codex_explicit_model_bypasses_routing` | Launch TUI with an explicit model and routing enabled | Completed task, no routing wrapper, normal exit and reopen | | `test_ug_claude_headless_prompt_argument`, `test_ug_claude_headless_prompt_stdin`, `test_ug_claude_headless_prompt_after_separator` | Run Claude from a script using each prompt form | Structured final answer contains the file value; exit zero; no routing | | `test_ug_codex_headless_prompt_argument`, `test_ug_codex_headless_prompt_stdin`, `test_ug_codex_headless_prompt_after_separator` | Run Codex from a script using each prompt form | Completed turn and final answer contain the file value; exit zero; no routing | | `test_ug_claude_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` with routing enabled | Real file task completes; no routing wrapper | @@ -45,10 +41,10 @@ All tests live directly in `integration/`; shared mechanics live in `utils/`. | `test_ug_status_in_fresh_home_is_unconfigured` | Request status before configure | Unconfigured status | | `test_ug_auth_without_configuration_explains_how_to_configure` | Request auth before configure | Actionable setup error and nonzero exit | -With both agents selected there are **43 live cases** (8 interactive TUI cases) +With both agents selected there are **39 live cases** (4 interactive TUI cases) and **3 installation checks**. Parametrization varies argument spelling or routing mode, never hides the agent/provider in the test name. Duplicate boot-only cases -were merged into the Databricks, first-prompt, and explicit-model TUI journeys. +are incorporated into the Databricks configuration TUI journeys. Generated-file cleanup and strict app-server stdout assertions remain enforced. Provider configuration tests include ug's normal validation. Other setup uses @@ -61,8 +57,8 @@ than consuming `uv.lock`. Use `--dependency PACKAGE==VERSION` or replay the arch dependency graph to reproduce a user's combination. Every relevant same-repository PR and push to `main` runs both smoke and the full CUJ suite. Smoke covers the Databricks Hosted configure/TUI and headless argument journeys for both agents, -in two parallel jobs. The full suite runs all 43 cases across eight parallel jobs: -Claude/Codex × configure, routing, headless, and other commands/lifecycle checks. +in two parallel jobs. The full suite runs all 39 cases across six parallel jobs: +Claude/Codex × configure, headless, and other commands/lifecycle checks. The `All integration tests` check requires every selected integration job to pass; full coverage does not depend on a label or a manual request. @@ -91,7 +87,7 @@ the gate. The descriptive jobs provide the actual coverage and diagnostics. | Provider switching, relayed/subscription MPS | Not covered by the four provider journeys | | TUI initial prompt supplied on the launch command line | Not yet covered; headless prompt arguments are covered | | Follow-up turns and conversation resume | Not covered; reopen proves startup, not conversation resume | -| Claude child model identity | Verified only when the actual start event reports it; missing fields remain unknown | +| Claude/Codex interactive smart routing | Deferred at the user's request; routing jobs and live journeys removed. Unit/component routing tests remain, but do not establish live routing behavior. | | Full allow/deny tool-permission matrix | Not covered; onboarding/trust uses actual TUI choices | | Desktop Codex app, Isaac itself, auto-upgrades | Not covered by command forwarding or pinned-version tests | | Native macOS/Windows managed settings, resize/signals | Separate platform coverage needed | diff --git a/tests/integration/README.md b/tests/integration/README.md index d204dfd05..48b57d9d8 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -43,7 +43,7 @@ for an npm mirror if public npm is unavailable. Older releases that only provide the `ucode` command require `--entry-point ucode`. Select one agent by providing only its version. Exact agent versions are -required; floating `latest`, caret, and tilde versions are rejected. Provider and default-routing CUJs use the workspace's configuration and need no model input. Cases that +required; floating `latest`, caret, and tilde versions are rejected. Hosted provider CUJs use the workspace's configuration and need no model input. Cases that exercise explicit model arguments use a real `system.ai` model already discovered by `ug configure`, recorded in that case's `model.json`. Optional `--claude-model` and `--codex-model` overrides reproduce a particular model-related failure. @@ -80,8 +80,6 @@ All user journeys are top-level tests. There is no separate regressions category ```text test_ug_configure_claude.py # Databricks Hosted and Anthropic MPS test_ug_configure_codex.py # Databricks Hosted and OpenAI MPS -test_smart_routing_claude.py # subagent, explicit model -test_smart_routing_codex.py # first prompt, explicit model test_ug_claude_headless.py # script prompts, models, caller settings test_ug_codex_headless.py # script prompts and model arguments test_ug_claude_commands.py # command help forwarding @@ -102,8 +100,8 @@ process/terminal mechanics, evidence, and cleanup. Fixtures supply an isolated session and credentials; none manufacture or configure application state. Provider CUJs use normal ug validation, then require their own completed -interactive task. Routing CUJs skip the preliminary validation prompt and require -the actual routed TUI task instead. Tests disable optional Databricks AI Tools and +interactive task. Other journeys skip preliminary validation when they provide +their own task or command assertions. Tests disable optional Databricks AI Tools and pass `--skip-upgrade` to preserve the selected version. They use real onboarding and trust choices, without seeded acceptance or disabled agent sandboxing. If a routed child asks to locate the random fixture beneath the disposable project, @@ -112,11 +110,10 @@ permission dialog; any broader permission request fails immediately. A fixture file contains an unpredictable value absent from the prompt. Success requires an assistant answer in the real agent transcript containing that value, -plus normal TUI exit. Codex evidence requires its task-complete event. Subagent -CUJs require a separate child transcript, child answer, and a correlated routing -decision. Claude uses its real child-start audit; its model is unknown when the -event omits it. Codex requires native parent linkage and a completed child turn -using the routed model, excluding inherited parent turns. +plus normal TUI exit. Codex evidence requires its task-complete event. +Interactive smart-routing journeys and their Claude/Codex CI shards are deferred +at the user's request. Unit/component routing tests remain; live first-prompt, +subagent routing, and interactive explicit-model bypass are not covered. MPS CUJs select the existing services already used by e2e: @@ -128,7 +125,7 @@ Use `--codex-provider-model` when that OpenAI service allows a different model. Those choices are recorded in `versions.json`. No service is created or modified. A missing service or permission fails the selected CUJ, rather than skipping it. -There are **43 live cases** (including 8 TUI journeys) and **3 installation +There are **39 live cases** (including 4 TUI journeys) and **3 installation checks** with both agents. See the named coverage and gaps matrix in [../README.md](../README.md). @@ -136,14 +133,14 @@ checks** with both agents. See the named coverage and gaps matrix in # Append one of these selections to the runner command: -- -m live # default: all live user journeys -- -m smoke # four Hosted configure/TUI and headless argument journeys --- -m tui # ten complete interactive TUI journeys +-- -m tui # four complete provider-configuration TUI journeys -- -k test_ug_codex_app_server_client_initializes # one named journey and its variants # Use --installation-only before -- for package checks without credentials. ``` The old focused checks are now descriptive CUJs with setup and outcomes visible -in each test. Duplicate boot-only checks are incorporated into the Databricks, -first-prompt, and explicit-model TUI journeys. Real failures, including generated +in each test. Duplicate boot-only checks are incorporated into the Databricks +configuration TUI journeys. Real failures, including generated config left after revert and banners on app-server stdout, remain assertions. MCP/skills functionality, tracing, the broad configure-option matrix, and other agents are outside this focused revision. @@ -196,13 +193,12 @@ workspace discovery inside each test; only explicit-model scenarios choose and record a discovered `system.ai` model as a test argument. Every relevant same-repository PR and push to `main` runs **Smoke journeys** and **Full journeys** concurrently. Smoke runs the Hosted configure/TUI and headless -argument journey for each agent (four cases, two agent jobs). Full runs all 43 -live cases, including those smoke cases, in eight disjoint shards: +argument journey for each agent (four cases, two agent jobs). Full runs all 39 +live cases, including those smoke cases, in six disjoint shards: | Group, per agent | Marker | Keyword filter | | --- | --- | --- | | configure | `live and tui` | `configure` | -| routing | `live and tui` | `not configure` | | headless | `live and not tui` | `headless` | | commands | `live and not tui` | `not headless` | @@ -225,7 +221,7 @@ The workflow consumes the stored bearer; it does not mint or refresh credentials For a manual run, use **Actions → Integration → Run workflow**, select the branch, and choose `full` (default), `smoke`, `tui`, or `installation`. `live` remains an alias for `full`. Manual subsets are explicit: `smoke` runs just the four smoke -cases; `tui` runs all ten TUI cases across the configure/routing shards. Installation +cases; `tui` runs all four TUI cases across the configure shards. Installation checks always run. Set the ug/agent versions. From the CLI: ```bash @@ -285,13 +281,13 @@ use the same OS/architecture as the original run; add `--platform linux/amd64` to both `docker build` and `docker run` on an ARM Mac to match GitHub's Ubuntu runner. Changing platforms or resolving a fresh npm lock is a new comparison, not an exact dependency replay. -Use `-- -m tui` for the eight interactive journeys or -`-- -k test_smart_routing_codex_first_prompt` to narrow a failure. Each rerun needs a new output directory. Inspect: +Use `-- -m tui` for the four interactive journeys or +`-- -k test_ug_configure_codex_databricks` to narrow a failure. Each rerun needs a new output directory. Inspect: - `junit.xml` for the failing case and assertion. - `artifacts//command-*.json` for the real argv, exit status, stdout and stderr. -- `artifacts//first-session.json`, `provider-session.json`, `first-prompt.json`, - `subagent-task.json`, or `reopen.json` for rendered terminal screens, raw terminal +- `artifacts//first-session.json`, `provider-session.json`, or `reopen.json` + for rendered terminal screens, raw terminal output, keyboard actions, exit status, routing logs, and actual agent-session records. - `install.log` for resolution/bootstrap failures. diff --git a/tests/integration/test_smart_routing_claude.py b/tests/integration/test_smart_routing_claude.py deleted file mode 100644 index 35dc32807..000000000 --- a/tests/integration/test_smart_routing_claude.py +++ /dev/null @@ -1,71 +0,0 @@ -"""CUJs: real child-task routing and explicit-model bypass in Claude's TUI.""" - -import pytest -from utils.evidence import FileTask, assert_subagent_routed -from utils.terminal import AgentTerminal - -pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.claude] - - -def test_smart_routing_claude_subagent(live_session, workspace): - """Scenario: ask Claude to delegate a file-reading task with routing enabled. - - Expected: a real child session has a correlated gateway routing decision, - the child returns the file value, and the parent returns that result. - Child model identity remains explicitly unknown if its event omits it. - """ - session = live_session - task = FileTask(session) - session.run( - "configure", - "--agents", - "claude", - "--workspaces", - workspace, - "--skip-validate", - "--skip-upgrade", - "--disable-databricks-ai-tools", - ) - - command = [str(session.binary), "claude", "--enable-smart-routing"] - with AgentTerminal(session, "claude", command, "subagent-task") as tui: - tui.boot() - tui.submit(task.delegate_prompt) - tui.wait_for_task(task, timeout=240) - task.assert_completed(session, "claude", child=True) - assert_subagent_routed(session, "claude", task) - tui.exit_normally() - task.assert_completed(session, "claude") - - -def test_smart_routing_claude_explicit_model_bypasses_routing(live_session, workspace): - """Scenario: request an explicit model while launching an interactive routing session. - - Expected: claude completes the real task with the caller's model choice, - starts no routing wrapper, exits normally, and reopens with usable input. - """ - session = live_session - task = FileTask(session) - session.run( - "configure", - "--agents", - "claude", - "--workspaces", - workspace, - "--skip-validate", - "--skip-upgrade", - "--disable-databricks-ai-tools", - ) - model = session.model_for_explicit_case("claude") - command = [str(session.binary), "claude", "--enable-smart-routing", "--model", model] - with AgentTerminal(session, "claude", command, "explicit-model") as tui: - tui.boot() - tui.submit(task.prompt) - tui.wait_for_task(task) - tui.exit_normally() - task.assert_completed(session, "claude") - session.assert_not_routed() - with AgentTerminal(session, "claude", command, "reopen") as tui: - tui.boot() - tui.check_input_and_exit() - session.assert_not_routed() diff --git a/tests/integration/test_smart_routing_codex.py b/tests/integration/test_smart_routing_codex.py deleted file mode 100644 index ec885c329..000000000 --- a/tests/integration/test_smart_routing_codex.py +++ /dev/null @@ -1,75 +0,0 @@ -"""CUJs: real first-prompt routing and explicit-model bypass in Codex's TUI.""" - -import pytest -from utils.evidence import FileTask -from utils.terminal import AgentTerminal - -pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.codex] - - -def test_smart_routing_codex_first_prompt(live_session, workspace): - """Scenario: enable smart routing and submit Codex's first interactive prompt. - - Expected: a real gateway decision selects a model and Codex completes the - file-reading task before a normal exit. A routing fallback does not pass. - """ - session = live_session - task = FileTask(session) - session.run( - "configure", - "--agents", - "codex", - "--workspaces", - workspace, - "--skip-validate", - "--skip-upgrade", - "--disable-databricks-ai-tools", - ) - - command = [str(session.binary), "codex", "--enable-smart-routing"] - with AgentTerminal(session, "codex", command, "first-prompt") as tui: - tui.boot() - assert "[ROUTE]" not in session.routing_log("codex"), "Boot must not route a prompt" - tui.submit(task.prompt) - tui.wait_for_task(task) - log = session.routing_log("codex") - assert "[ROUTE] selected" in log, log - assert "selection failed" not in log and "settings/update timed out" not in log, log - tui.exit_normally() - task.assert_completed(session, "codex") - with AgentTerminal(session, "codex", command, "reopen") as tui: - tui.boot() - tui.check_input_and_exit() - - -def test_smart_routing_codex_explicit_model_bypasses_routing(live_session, workspace): - """Scenario: request an explicit model while launching an interactive routing session. - - Expected: codex completes the real task with the caller's model choice, - starts no routing wrapper, exits normally, and reopens with usable input. - """ - session = live_session - task = FileTask(session) - session.run( - "configure", - "--agents", - "codex", - "--workspaces", - workspace, - "--skip-validate", - "--skip-upgrade", - "--disable-databricks-ai-tools", - ) - model = session.model_for_explicit_case("codex") - command = [str(session.binary), "codex", "--enable-smart-routing", "--", "--model", model] - with AgentTerminal(session, "codex", command, "explicit-model") as tui: - tui.boot() - tui.submit(task.prompt) - tui.wait_for_task(task) - tui.exit_normally() - task.assert_completed(session, "codex") - session.assert_not_routed() - with AgentTerminal(session, "codex", command, "reopen") as tui: - tui.boot() - tui.check_input_and_exit() - session.assert_not_routed() From 53c97ca9931c203056cdbfb44d9f0a3a2af8cdc4 Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Mon, 14 Sep 2026 21:30:36 -0400 Subject: [PATCH 9/9] Fix Codex model coverage and gate full integration checks --- .github/workflows/ci.yml | 20 ++++++- .github/workflows/integration.yml | 13 ++--- tests/README.md | 13 +++-- tests/integration/README.md | 91 ++++++++++++++++++++++++++--- tests/integration/utils/evidence.py | 12 ++++ tests/integration/utils/terminal.py | 11 +--- tests/test_e2e.py | 6 +- tests/test_integration_evidence.py | 31 ++++++++++ 8 files changed, 164 insertions(+), 33 deletions(-) create mode 100644 tests/test_integration_evidence.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index e7d45d7b7..4a2710734 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -9,6 +9,10 @@ on: permissions: contents: read +concurrency: + group: ci-${{ github.event.pull_request.number || github.ref }}-${{ github.event_name }} + cancel-in-progress: true + jobs: test: name: Unit tests @@ -113,6 +117,15 @@ jobs: RESULT: ${{ needs.e2e-shards.result }} run: test "$RESULT" = success + integration: + name: Integration + # These suites use the same live workspace. Let agent launch coverage + # finish before starting integration requests, even if an agent failed. + if: ${{ !cancelled() }} + needs: e2e + uses: ./.github/workflows/integration.yml + secrets: inherit + # Preserve the exact contexts required by the repository's branch rules. # Keep these aliases until the rules and all active branches migrate together. required-tests: @@ -130,11 +143,12 @@ jobs: required-e2e: name: e2e if: ${{ !cancelled() }} - needs: e2e + needs: [e2e, integration] runs-on: ubuntu-latest timeout-minutes: 5 steps: - - name: Require all agent tests to pass + - name: Require all agent and integration tests to pass env: RESULT: ${{ needs.e2e.result }} - run: test "$RESULT" = success + INTEGRATION_RESULT: ${{ needs.integration.result }} + run: test "$RESULT" = success && test "$INTEGRATION_RESULT" = success diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index 7914b6de8..604db3764 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -1,11 +1,7 @@ name: Integration on: - pull_request: - paths: ['src/**', 'tests/**', 'scripts/run_integration.py', 'pyproject.toml', 'uv.lock', '.github/workflows/integration.yml'] - push: - branches: [main] - paths: ['src/**', 'tests/**', 'scripts/run_integration.py', 'pyproject.toml', 'uv.lock', '.github/workflows/integration.yml'] + workflow_call: workflow_dispatch: inputs: suite: @@ -170,14 +166,17 @@ jobs: full: name: Full journeys · ${{ matrix.agent == 'claude' && 'Claude' || 'Codex' }} · ${{ matrix.group == 'configure' && 'Configure' || matrix.group == 'headless' && 'Headless' || 'Commands' }} - if: ${{ inputs.suite != 'smoke' }} - needs: workspace + if: ${{ !cancelled() && inputs.suite != 'smoke' && needs.workspace.result == 'success' }} + needs: [workspace, smoke] runs-on: ubuntu-22.04 # The runner enforces its own 3600s pytest deadline; keep the job above it # so a slow run fails through the runner's report instead of a job kill. timeout-minutes: 70 strategy: fail-fast: false + # All shards share one workspace/model quota. Serial execution retains + # every case without a burst of overlapping model requests. + max-parallel: 1 matrix: agent: [claude, codex] group: ${{ fromJSON(inputs.suite == 'tui' && '["configure"]' || '["configure", "headless", "commands"]') }} diff --git a/tests/README.md b/tests/README.md index c64a68d39..b7645b27a 100644 --- a/tests/README.md +++ b/tests/README.md @@ -57,8 +57,10 @@ than consuming `uv.lock`. Use `--dependency PACKAGE==VERSION` or replay the arch dependency graph to reproduce a user's combination. Every relevant same-repository PR and push to `main` runs both smoke and the full CUJ suite. Smoke covers the Databricks Hosted configure/TUI and headless argument journeys for both agents, -in two parallel jobs. The full suite runs all 39 cases across six parallel jobs: -Claude/Codex × configure, headless, and other commands/lifecycle checks. +in two parallel jobs. After smoke finishes, the full suite runs all 39 cases +across six serial jobs: Claude/Codex × configure, headless, and other +commands/lifecycle checks. CI calls integration after the existing e2e shards +finish, including when an e2e shard fails, to avoid overlapping their model load. The `All integration tests` check requires every selected integration job to pass; full coverage does not depend on a label or a manual request. @@ -74,9 +76,10 @@ Unit tests still run as one job. Both matrices use `fail-fast: false` so one failure does not cancel other coverage. The small `test` and `e2e` compatibility gates retain the exact status contexts -required by the repository's branch rules. They pass only when `Unit tests` and -`All agent tests`, respectively, succeed; a failed or skipped dependency fails -the gate. The descriptive jobs provide the actual coverage and diagnostics. +required by the repository's branch rules. `test` requires `Unit tests`; `e2e` +requires both `All agent tests` and the complete integration workflow. A failed +or skipped dependency fails the gate, and a running integration suite keeps it +pending. The descriptive jobs provide the actual coverage and diagnostics. ## Gaps and deferred scope diff --git a/tests/integration/README.md b/tests/integration/README.md index 48b57d9d8..e3b5bf159 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -17,7 +17,7 @@ Pytest and the PTY/screen libraries (pexpect and pyte) live in a different virtu missing application dependency. No packages are installed into your existing agent installations or checkout's `.venv`. Native live runs refuse existing machine-wide Claude/Codex configuration, which -could override the selected workspace even with a fresh home. Use the container +could override the selected workspace even with a fresh home. Use a clean VM in that case; the runner never edits or bypasses those managed settings. Use the existing e2e workspace and its `DATABRICKS_BEARER` credential. Locally, @@ -177,7 +177,8 @@ selection to installation checks, including when additional filters are used. ## Run in GitHub Actions -The **Integration** workflow runs on relevant pull requests and pushes to `main`. +The **CI** workflow calls **Integration** on pull requests and pushes to `main`, +after its existing e2e shards finish (even if one fails). It runs directly on fresh GitHub Ubuntu VMs, not inside the optional Docker image. Local native runs use the same runner; Colima/Docker provides a separate Linux container option. Matching dependency versions does not make those OS environments identical. @@ -191,8 +192,8 @@ is accepted). It never changes the secret or switches workspaces. There is no CI model-discovery or model-selection job. Real `ug configure` performs its normal workspace discovery inside each test; only explicit-model scenarios choose and record a discovered `system.ai` model as a test argument. -Every relevant same-repository PR and push to `main` runs **Smoke journeys** and -**Full journeys** concurrently. Smoke runs the Hosted configure/TUI and headless +Every same-repository PR and push to `main` runs **Smoke journeys**, followed by +**Full journeys** even if smoke fails. Smoke runs the Hosted configure/TUI and headless argument journey for each agent (four cases, two agent jobs). Full runs all 39 live cases, including those smoke cases, in six disjoint shards: @@ -205,10 +206,14 @@ live cases, including those smoke cases, in six disjoint shards: Each shard also selects `claude` or `codex` and installs only that CLI. The commands group includes lifecycle and app-server journeys. Cases remain serial inside each fresh VM because configure/revert can touch machine-level settings; -parallel runners isolate those writes as well as the PTYs. Both matrices use +separate runners isolate those writes as well as the PTYs. The six full shards +run one at a time to avoid bursts against the shared workspace/model quota. +Both matrices use `fail-fast: false` and upload uniquely named evidence even when another shard fails. The **All integration tests** check requires installation, workspace validation, smoke, and -all full shards to pass. Full coverage on PRs needs no label or opt-in. +all full shards to pass. The existing required `e2e` context also waits for the +complete integration workflow, so integration cannot still be running when +that gate passes. Full coverage on PRs needs no label or opt-in. Each job uses fresh consumer dependency resolution. There is no default dependency matrix. Manual dispatch accepts an @@ -233,7 +238,7 @@ gh run watch RUN_ID -R databricks/unity-gateway --exit-status ``` GitHub enables manual dispatch once the workflow exists on the default branch. -Before this PR merges, its pull-request event runs the workflow. Missing +Before this PR merges, CI's pull-request event calls the workflow. Missing credentials or a workspace mismatch fail the workspace job. Expired or invalid credentials fail the actual workspace calls. Those failures do not count as live test passes. @@ -296,9 +301,19 @@ selectors only after confirming the agent's intended UI changed; do not seed its onboarding state or relax the prompt/task assertions. Test homes are deleted after each case; redacted diagnostics remain. For manual interaction, configure a fresh home with the same installed binaries and recorded public CLI arguments. +Definitive API errors and the client's exhausted retry limit fail the TUI wait +immediately with the actual screen. A transient 429/503 while the client is still +retrying is not treated as terminal; the suite adds no task retries of its own. ## Colima / Docker +Docker is an optional installation/command-check environment, not a guaranteed +replacement for the native Linux live suite. Default Colima/Docker security +policies can reject Codex's user namespaces; the image also lacks `sudo` needed +by machine-level configure/revert journeys. Do not disable the agent sandbox or +use privileged/unconfined containers to turn these into passing results. Use a +clean Ubuntu 22.04 VM for the complete live run below. + Colima provides the Linux Docker engine on macOS. The optional image pins the Python, Node, uv, and Databricks toolchain; the same runner selects ug and agent versions inside it. Build from the repository root: @@ -318,7 +333,7 @@ docker run --rm --init \ ug-integration \ --ug-version YOUR_RELEASE_VERSION \ --claude-version 2.1.268 --codex-version 0.154.0 \ - -- -m live + --installation-only ``` Use a new results volume for each run, or pass a new `--output /results/NAME`. @@ -330,3 +345,63 @@ native runs also depend on the host's OS and toolchain. The explicit build archive includes only the runner and integration files, even with legacy Docker builders that ignore per-Dockerfile ignore rules. It also omits macOS extended attributes that Linux cannot unpack. + +### Complete local checkout run on the Databricks network + +Run from the checkout root in Bash on a clean Ubuntu 22.04 machine/VM with +Python 3.12, uv 0.9.8, Node 22.19.0/npm, Databricks CLI 1.9.0, `sudo`, and +`bubblewrap`. Install bubblewrap with `sudo apt-get install bubblewrap`. Verify +the normal sandbox before starting: + +```bash +bwrap --ro-bind / / --unshare-user --proc /proc --dev /dev /usr/bin/true +``` + +On macOS, Lima can supply a separate VM with no host mounts: + +```bash +limactl start --name ug-integration --plain --cpus 2 --memory 4 --disk 12 \ + --yes template:ubuntu-22.04 +limactl shell ug-integration +``` + +Install the prerequisites and copy/clone the checkout inside that VM. On recent +Apple Silicon, the original Ubuntu 22.04 ARM kernel can crash `cryptography` +with `Illegal instruction` before ug starts. Install Ubuntu's supported +`linux-generic-hwe-22.04` package and restart the VM. Do not work around it by +altering Python dependencies or weakening tests. Linux/ARM is not an exact +replay of GitHub's Linux/AMD64 environment; the run records its actual platform. + +Use your authorized login and explicitly selected profile; never download CI +secrets. The runner builds the checkout wheel and isolates the installed agents, +ug, test dependencies, and per-case homes: + +```bash +set -euo pipefail +integration_profile=eng-ml-inference-team-us-east-1 +integration_workspace=https://eng-ml-inference-team-us-east-1.cloud.databricks.com +integration_index=https://pypi-proxy.cloud.databricks.com/simple +integration_registry=https://npm-proxy.cloud.databricks.com/ + +# Keep this setting on both login and token retrieval when using plaintext storage. +export DATABRICKS_AUTH_STORAGE=plaintext +databricks auth login --host "$integration_workspace" --profile "$integration_profile" +export DATABRICKS_BEARER +DATABRICKS_BEARER=$(databricks auth token --host "$integration_workspace" \ + --profile "$integration_profile" --output json | jq -er '.access_token | select(length > 0)') +uv run --no-project --python 3.12 python scripts/run_integration.py \ + --python 3.12 --ug-version checkout --workspace "$integration_workspace" \ + --default-index "$integration_index" --npm-registry "$integration_registry" \ + --claude-version 2.1.268 --codex-version 0.154.0 -- -m live +unset DATABRICKS_BEARER +``` + +This runs all 39 live cases. For the three installation checks, run the same +runner/version/index arguments with `--installation-only` and omit `-- -m live`; +no bearer or workspace is needed. Results remain under `.integration-runs/`. +Each invocation needs a new output directory; an existing one is rejected. +The runner returns nonzero on installation or test failure. + +Do not drop the mirror flags if public registries resolve to `127.0.0.1` or return +`ECONNREFUSED`. The runner deliberately ignores host `.npmrc` and resolver settings. +Outside the Databricks network, use reachable package indexes explicitly instead. diff --git a/tests/integration/utils/evidence.py b/tests/integration/utils/evidence.py index 6cfafdac7..052b75442 100644 --- a/tests/integration/utils/evidence.py +++ b/tests/integration/utils/evidence.py @@ -1,10 +1,22 @@ """Read real agent transcripts and ug routing records without changing them.""" import json +import re import uuid from pathlib import Path +def assert_no_terminal_api_error(screen: str) -> None: + """Fail on definitive client errors, not an in-progress transient retry.""" + error = re.search( + r"unexpected status (?:400|401|403|404|405|409|422)\b|PERMISSION_DENIED" + r"|exceeded retry limit", + screen, + re.IGNORECASE, + ) + assert error is None, "Agent returned a terminal API error:\n" + screen + + def read_jsonl(path: Path) -> list[dict]: if not path.is_file(): return [] diff --git a/tests/integration/utils/terminal.py b/tests/integration/utils/terminal.py index a47290807..1a68b72cd 100644 --- a/tests/integration/utils/terminal.py +++ b/tests/integration/utils/terminal.py @@ -16,12 +16,7 @@ import pexpect import pyte -from .evidence import agent_sessions - -NON_RETRYABLE_AGENT_ERROR = re.compile( - r"unexpected status (?:400|401|403|404|405|409|422)\b|PERMISSION_DENIED", - re.IGNORECASE, -) +from .evidence import agent_sessions, assert_no_terminal_api_error class TerminalScreen(pyte.Screen): @@ -291,9 +286,7 @@ def wait_for_task(self, task, timeout=180): def completed(screen): nonlocal permission_in_progress - assert not NON_RETRYABLE_AGENT_ERROR.search(screen), ( - "Agent returned a non-retryable API error:\n" + screen - ) + assert_no_terminal_api_error(screen) if "Do you want to proceed?" in screen: if permission_in_progress: return False diff --git a/tests/test_e2e.py b/tests/test_e2e.py index a562a46b2..a7199ce23 100644 --- a/tests/test_e2e.py +++ b/tests/test_e2e.py @@ -498,7 +498,11 @@ def test_launch_codex_per_model(self, tmp_path, monkeypatch, e2e_state, e2e_work failures.append(f"model={model} timed out after {timeout_seconds}s") continue - if result.returncode != 0 or not (result.stdout or result.stderr).strip(): + if ( + result.returncode != 0 + or not result.stdout.strip() + or f"model: {codex.codex_model_id(model)}\n" not in result.stderr + ): # Keep a generous tail of stderr. codex-cli logs a non-fatal model-listing error # first and the actual cause last, so a short prefix reports the wrong problem — # at 200 chars the geography failure above read as a `/v1/models` routing error. diff --git a/tests/test_integration_evidence.py b/tests/test_integration_evidence.py new file mode 100644 index 000000000..e057a2062 --- /dev/null +++ b/tests/test_integration_evidence.py @@ -0,0 +1,31 @@ +"""Unit checks for interpreting terminal evidence, not live agent substitutes.""" + +import pytest + +from tests.integration.utils.evidence import assert_no_terminal_api_error + + +@pytest.mark.parametrize( + "screen", + [ + "■ exceeded retry limit, last status: 429 Too Many Requests", + "■ unexpected status 403 Forbidden: PERMISSION_DENIED", + "■ unexpected status 401 Unauthorized", + ], +) +def test_terminal_api_failure_reports_the_actual_error(screen): + with pytest.raises(AssertionError, match="Agent returned a terminal API error") as error: + assert_no_terminal_api_error(screen) + assert screen in str(error.value) + + +@pytest.mark.parametrize( + "screen", + [ + "Reconnecting... 1/5 (unexpected status 429 Too Many Requests)", + "Reconnecting... 1/5 (unexpected status 503 Service Unavailable)", + "Working (5s · esc to interrupt)", + ], +) +def test_transient_retries_and_running_tasks_are_not_terminal_errors(screen): + assert_no_terminal_api_error(screen)