diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a86cc5913..4a2710734 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -9,8 +9,13 @@ on: permissions: contents: read +concurrency: + group: ci-${{ github.event.pull_request.number || github.ref }}-${{ github.event_name }} + cancel-in-progress: true + jobs: test: + name: Unit tests runs-on: ubuntu-latest env: # uv.lock pins packages to Databricks' internal pypi-proxy, which hosted @@ -23,8 +28,41 @@ jobs: - run: uv lock - run: uv run pytest --ignore=tests/test_e2e.py --ignore=tests/test_e2e_tracing.py - e2e: + e2e-shards: + name: ${{ matrix.name }} runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + include: + - group: gateway + name: Gateway API tests + keyword: not (TestCodexLaunch or TestClaudeLaunch or TestConfigureSubset or TestModelProviderLaunch or TestGeminiLaunch or TestGeminiFreshInstall or TestGeminiAuthRecovery or TestOpencodeLaunch or TestCopilotLaunch or TestPiLaunch) + package: '' + - group: claude + name: Agent launch tests · Claude + keyword: TestClaudeLaunch or TestConfigureSubset or (TestModelProviderLaunch and not codex) + package: '@anthropic-ai/claude-code' + - group: codex + name: Agent launch tests · Codex + keyword: TestCodexLaunch or (TestModelProviderLaunch and codex) + package: '@openai/codex' + - group: gemini + name: Agent launch tests · Gemini + keyword: TestGeminiLaunch or TestGeminiFreshInstall or TestGeminiAuthRecovery + package: '@google/gemini-cli' + - group: opencode + name: Agent launch tests · OpenCode + keyword: TestOpencodeLaunch + package: opencode-ai@1 + - group: copilot + name: Agent launch tests · Copilot + keyword: TestCopilotLaunch + package: '@github/copilot@1.0.80' + - group: pi + name: Agent launch tests · Pi + keyword: TestPiLaunch + package: '@earendil-works/pi-coding-agent' env: # See the test job: hosted runners can't reach the internal pypi-proxy, # so resolve against public PyPI (paired with the `uv lock` step below). @@ -42,34 +80,75 @@ jobs: DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} # Subscription OAuth token for the relayed-launch e2e test; the test skips when unset. CLAUDE_CODE_OAUTH_TOKEN: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }} + AGENT_PACKAGE: ${{ matrix.package }} + TEST_KEYWORD: ${{ matrix.keyword }} steps: - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 - uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0 - uses: databricks/setup-cli@bdb89f81c11a5bd647fd55b585b7c396ec68a25a # v1.0.0 - # The agent launch tests `_require_binary("codex")` etc. and skip when - # the CLI isn't on PATH. Install all six so each TestXxxLaunch test - # actually runs instead of skipping. - # Pin Copilot because 1.0.81-6 sends unsupported parameters through BYOK providers. - - name: Install agent CLIs - run: npm install -g - @anthropic-ai/claude-code - @openai/codex - @google/gemini-cli - opencode-ai@1 - @github/copilot@1.0.80 - @earendil-works/pi-coding-agent + - name: Install the shard's agent CLI + if: ${{ matrix.package != '' }} + run: npm install -g "$AGENT_PACKAGE" - run: uv lock - run: uv tool install . # Redirect stdin so any interactive `databricks auth login --no-browser` # fallback EOFs instead of hanging the runner. With DATABRICKS_BEARER # set, the auth code path doesn't shell out at all — this is a safety # net for any code path we may have missed. - - run: uv run pytest tests/test_e2e.py -v < /dev/null + - run: uv run pytest tests/test_e2e.py -v -k "$TEST_KEYWORD" < /dev/null # MLflow tracing e2e lives in its own file and needs the `tracing` # extra so `import mlflow` resolves (otherwise the test importorskips # and silently passes as skipped). `!cancelled()` lets this run even # when the previous pytest step failed — the two suites are # independent and one shouldn't mask the other. - - if: ${{ !cancelled() }} + - if: ${{ !cancelled() && matrix.group == 'claude' }} run: uv run --extra tracing pytest tests/test_e2e_tracing.py -v < /dev/null + + e2e: + name: All agent tests + if: ${{ !cancelled() }} + needs: e2e-shards + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - name: Require every e2e shard to pass + env: + RESULT: ${{ needs.e2e-shards.result }} + run: test "$RESULT" = success + + integration: + name: Integration + # These suites use the same live workspace. Let agent launch coverage + # finish before starting integration requests, even if an agent failed. + if: ${{ !cancelled() }} + needs: e2e + uses: ./.github/workflows/integration.yml + secrets: inherit + + # Preserve the exact contexts required by the repository's branch rules. + # Keep these aliases until the rules and all active branches migrate together. + required-tests: + name: test + if: ${{ !cancelled() }} + needs: test + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - name: Require unit tests to pass + env: + RESULT: ${{ needs.test.result }} + run: test "$RESULT" = success + + required-e2e: + name: e2e + if: ${{ !cancelled() }} + needs: [e2e, integration] + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - name: Require all agent and integration tests to pass + env: + RESULT: ${{ needs.e2e.result }} + INTEGRATION_RESULT: ${{ needs.integration.result }} + run: test "$RESULT" = success && test "$INTEGRATION_RESULT" = success diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml new file mode 100644 index 000000000..604db3764 --- /dev/null +++ b/.github/workflows/integration.yml @@ -0,0 +1,221 @@ +name: Integration + +on: + workflow_call: + workflow_dispatch: + inputs: + suite: + description: Which integration checks to run + type: choice + options: [full, smoke, live, tui, installation] + default: full + ug_version: + description: Exact ug release, or checkout + default: checkout + required: true + entry_point: + description: Console command (older releases may only have ucode) + type: choice + options: [ug, ucode] + default: ug + claude_version: + description: Exact Claude Code version + default: 2.1.268 + required: true + codex_version: + description: Exact Codex version + default: 0.154.0 + required: true + dependency: + description: Optional exact dependency for reproduction (e.g. tomlkit==0.14.0) + default: '' + required: false + default_index: + description: Python package index containing the requested ug version + default: https://pypi.org/simple + required: true + +permissions: + contents: read + +concurrency: + group: integration-${{ github.event.pull_request.number || github.ref }}-${{ github.event_name }} + cancel-in-progress: true + +env: + EXPECTED_WORKSPACE: https://eng-ml-inference-team-us-east-1.cloud.databricks.com + UG_VERSION: ${{ inputs.ug_version || 'checkout' }} + ENTRY_POINT: ${{ inputs.entry_point || 'ug' }} + CLAUDE_VERSION: ${{ inputs.claude_version || '2.1.268' }} + CODEX_VERSION: ${{ inputs.codex_version || '0.154.0' }} + PACKAGE_INDEX: ${{ inputs.default_index || 'https://pypi.org/simple' }} + +jobs: + installation: + name: Installation tests + runs-on: ubuntu-22.04 + timeout-minutes: 20 + steps: + - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + with: + fetch-depth: 0 + - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 + with: + node-version: 22.19.0 + - uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0 + with: + version: 0.9.8 + - name: Test a fresh installed package without credentials + run: | + uv run --no-project --python 3.12 python scripts/run_integration.py \ + --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ + --claude-version "$CLAUDE_VERSION" --codex-version "$CODEX_VERSION" \ + --default-index "$PACKAGE_INDEX" --installation-only --output "$RUNNER_TEMP/ug-integration" + - name: Upload installation evidence + if: ${{ !cancelled() }} + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 + with: + name: integration-installation + include-hidden-files: true + path: | + ${{ runner.temp }}/ug-integration/*.json + ${{ runner.temp }}/ug-integration/*.txt + ${{ runner.temp }}/ug-integration/*.xml + ${{ runner.temp }}/ug-integration/*.log + ${{ runner.temp }}/ug-integration/artifacts/ + ${{ runner.temp }}/ug-integration/wheels/ + + workspace: + name: Workspace validation + if: ${{ inputs.suite != 'installation' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) }} + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - name: Verify the existing e2e workspace and credential are configured + env: + UCODE_TEST_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} + DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} + run: | + python3 - <<'PY' + import os + expected = os.environ["EXPECTED_WORKSPACE"].rstrip("/") + actual = os.environ["UCODE_TEST_WORKSPACE"].strip().rstrip("/") + if actual != expected: + raise SystemExit("UCODE_TEST_WORKSPACE does not match the expected CI workspace.") + if not os.environ["DATABRICKS_BEARER"].strip(): + raise SystemExit("The existing e2e DATABRICKS_BEARER secret is missing.") + print("CI workspace matches; the existing e2e credential is present.") + PY + + smoke: + name: Smoke journeys · ${{ matrix.agent == 'claude' && 'Claude' || 'Codex' }} + if: ${{ inputs.suite != 'tui' }} + needs: workspace + runs-on: ubuntu-22.04 + timeout-minutes: 20 + strategy: + fail-fast: false + matrix: + agent: [claude, codex] + env: + UCODE_TEST_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} + DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} + DEPENDENCY: ${{ inputs.dependency }} + AGENT: ${{ matrix.agent }} + TEST_MARKER: live and smoke and ${{ matrix.agent }} + TEST_KEYWORD: '' + ARTIFACT_NAME: integration-smoke-${{ matrix.agent }} + steps: &live-steps + - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + with: + fetch-depth: 0 + - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 + with: + node-version: 22.19.0 + - uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0 + with: + version: 0.9.8 + - uses: databricks/setup-cli@bdb89f81c11a5bd647fd55b585b7c396ec68a25a # v1.0.0 + - name: Run the selected live cases against the e2e workspace + shell: bash + run: | + case "$AGENT" in + claude) args=(--claude-version "$CLAUDE_VERSION") ;; + codex) args=(--codex-version "$CODEX_VERSION") ;; + esac + if [[ -n "$DEPENDENCY" ]]; then + args+=(--dependency "$DEPENDENCY") + fi + uv run --no-project --python 3.12 python scripts/run_integration.py \ + --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ + --default-index "$PACKAGE_INDEX" --output "$RUNNER_TEMP/ug-integration" \ + "${args[@]}" -- -m "$TEST_MARKER" -k "$TEST_KEYWORD" + - name: Upload live test evidence + if: ${{ !cancelled() }} + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 + with: + name: ${{ env.ARTIFACT_NAME }} + include-hidden-files: true + path: | + ${{ runner.temp }}/ug-integration/*.json + ${{ runner.temp }}/ug-integration/*.txt + ${{ runner.temp }}/ug-integration/*.xml + ${{ runner.temp }}/ug-integration/*.log + ${{ runner.temp }}/ug-integration/artifacts/ + ${{ runner.temp }}/ug-integration/wheels/ + + full: + name: Full journeys · ${{ matrix.agent == 'claude' && 'Claude' || 'Codex' }} · ${{ matrix.group == 'configure' && 'Configure' || matrix.group == 'headless' && 'Headless' || 'Commands' }} + if: ${{ !cancelled() && inputs.suite != 'smoke' && needs.workspace.result == 'success' }} + needs: [workspace, smoke] + runs-on: ubuntu-22.04 + # The runner enforces its own 3600s pytest deadline; keep the job above it + # so a slow run fails through the runner's report instead of a job kill. + timeout-minutes: 70 + strategy: + fail-fast: false + # All shards share one workspace/model quota. Serial execution retains + # every case without a burst of overlapping model requests. + max-parallel: 1 + matrix: + agent: [claude, codex] + group: ${{ fromJSON(inputs.suite == 'tui' && '["configure"]' || '["configure", "headless", "commands"]') }} + env: + UCODE_TEST_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} + DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} + DEPENDENCY: ${{ inputs.dependency }} + AGENT: ${{ matrix.agent }} + TEST_MARKER: ${{ matrix.group == 'configure' && 'live and tui' || 'live and not tui' }} and ${{ matrix.agent }} + TEST_KEYWORD: ${{ matrix.group == 'configure' && 'configure' || matrix.group == 'headless' && 'headless' || 'not headless' }} + ARTIFACT_NAME: integration-full-${{ matrix.agent }}-${{ matrix.group }} + steps: *live-steps + + cujs: + name: All integration tests + if: ${{ !cancelled() && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) }} + needs: [installation, workspace, smoke, full] + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - name: Require every selected integration job to pass + env: + RESULTS: ${{ toJSON(needs) }} + SUITE: ${{ inputs.suite || 'full' }} + run: | + python3 - <<'PY' + import json + import os + results = json.loads(os.environ["RESULTS"]) + suite = os.environ["SUITE"] + required = ["installation"] + if suite != "installation": + required.append("workspace") + if suite != "tui": + required.append("smoke") + if suite != "smoke": + required.append("full") + failed = [job for job in required if results[job]["result"] != "success"] + if failed: + raise SystemExit("Integration jobs did not pass: " + ", ".join(failed)) + print("All selected integration jobs passed: " + ", ".join(required)) + PY diff --git a/scripts/run_integration.py b/scripts/run_integration.py index 9826437ce..2ac5f0c25 100644 --- a/scripts/run_integration.py +++ b/scripts/run_integration.py @@ -57,9 +57,7 @@ def exact_npm_version(value: str) -> str: def arguments(): parser = argparse.ArgumentParser(description=__doc__) source = parser.add_mutually_exclusive_group() - source.add_argument( - "--ug-version", default="checkout", help="Exact ucode release, or checkout." - ) + source.add_argument("--ug-version", default="checkout", help="Exact ug release, or checkout.") source.add_argument( "--ug-wheel", type=Path, help="Previously built wheel to reproduce a release." ) @@ -78,6 +76,11 @@ def arguments(): default="main.ucode.ci_openai_mps", help="Existing OpenAI MPS selected in the configure CUJ.", ) + parser.add_argument( + "--codex-provider-model", + default="gpt-5-nano", + help="Model allowed by the OpenAI MPS selected in the configure CUJ.", + ) parser.add_argument("--python", default=sys.executable, help="Python 3.12+ path or uv version.") parser.add_argument("--dependency", action="append", default=[], metavar="PACKAGE==VERSION") parser.add_argument("--constraints", type=Path, help="Replay a previous dependencies.txt.") @@ -238,6 +241,7 @@ def run(command, *, cwd=output, env=base_env, timeout=600) -> str: "codex_model": args.codex_model, "claude_provider": args.claude_provider, "codex_provider": args.codex_provider, + "codex_provider_model": args.codex_provider_model, "dependencies": args.dependency, "workspace": args.workspace, }, @@ -461,6 +465,7 @@ def run(command, *, cwd=output, env=base_env, timeout=600) -> str: "UG_INTEGRATION_AGENTS": ",".join(agents), "UG_INTEGRATION_CLAUDE_PROVIDER": args.claude_provider, "UG_INTEGRATION_CODEX_PROVIDER": args.codex_provider, + "UG_INTEGRATION_CODEX_PROVIDER_MODEL": args.codex_provider_model, "UCODE_TEST_WORKSPACE": args.workspace or "", "DATABRICKS_BEARER": bearer, } diff --git a/tests/README.md b/tests/README.md index 6a28cc363..b7645b27a 100644 --- a/tests/README.md +++ b/tests/README.md @@ -25,12 +25,6 @@ All tests live directly in `integration/`; shared mechanics live in `utils/`. | `test_ug_configure_claude_anthropic_mps` | Select Anthropic MPS in the real configure picker; launch Claude | Saved provider in status; completed TUI file task; normal exit | | `test_ug_configure_codex_databricks` | Configure Databricks Hosted; open Codex TUI and read a file | Completed assistant answer contains the file value; normal exit and reopen | | `test_ug_configure_codex_openai_mps` | Select OpenAI MPS in the real configure picker; launch Codex | Saved provider in status; completed TUI file task; normal exit | -| `test_smart_routing_claude_first_prompt` | Enable routing and type the first TUI prompt | Real routing decision and replay; completed task; no routing before submission; reopen | -| `test_smart_routing_codex_first_prompt` | Enable routing and type the first TUI prompt | Real routing decision; completed task; no fallback or routing before submission; reopen | -| `test_smart_routing_claude_subagent` | Delegate a file-reading task | Child answer, correlated routing decision/child start, parent answer | -| `test_smart_routing_codex_subagent` | Delegate a file-reading task | Native child-parent linkage; completed child turn uses the routed model; parent answer | -| `test_smart_routing_claude_explicit_model_bypasses_routing` | Launch TUI with an explicit model and routing enabled | Completed task, no routing wrapper, normal exit and reopen | -| `test_smart_routing_codex_explicit_model_bypasses_routing` | Launch TUI with an explicit model and routing enabled | Completed task, no routing wrapper, normal exit and reopen | | `test_ug_claude_headless_prompt_argument`, `test_ug_claude_headless_prompt_stdin`, `test_ug_claude_headless_prompt_after_separator` | Run Claude from a script using each prompt form | Structured final answer contains the file value; exit zero; no routing | | `test_ug_codex_headless_prompt_argument`, `test_ug_codex_headless_prompt_stdin`, `test_ug_codex_headless_prompt_after_separator` | Run Codex from a script using each prompt form | Completed turn and final answer contain the file value; exit zero; no routing | | `test_ug_claude_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` with routing enabled | Real file task completes; no routing wrapper | @@ -47,10 +41,10 @@ All tests live directly in `integration/`; shared mechanics live in `utils/`. | `test_ug_status_in_fresh_home_is_unconfigured` | Request status before configure | Unconfigured status | | `test_ug_auth_without_configuration_explains_how_to_configure` | Request auth before configure | Actionable setup error and nonzero exit | -With both agents selected there are **45 live cases** (10 interactive TUI cases) +With both agents selected there are **39 live cases** (4 interactive TUI cases) and **3 installation checks**. Parametrization varies argument spelling or routing mode, never hides the agent/provider in the test name. Duplicate boot-only cases -were merged into the Databricks, first-prompt, and explicit-model TUI journeys. +are incorporated into the Databricks configuration TUI journeys. Generated-file cleanup and strict app-server stdout assertions remain enforced. Provider configuration tests include ug's normal validation. Other setup uses @@ -60,11 +54,32 @@ optional Databricks AI Tools. Help forwarding does not claim MCP functionality. Fresh consumer dependency resolution covers the install path behind #496, rather than consuming `uv.lock`. Use `--dependency PACKAGE==VERSION` or replay the archived -dependency graph to reproduce a user's combination. +dependency graph to reproduce a user's combination. Every relevant same-repository +PR and push to `main` runs both smoke and the full CUJ suite. Smoke covers the +Databricks Hosted configure/TUI and headless argument journeys for both agents, +in two parallel jobs. After smoke finishes, the full suite runs all 39 cases +across six serial jobs: Claude/Codex × configure, headless, and other +commands/lifecycle checks. CI calls integration after the existing e2e shards +finish, including when an e2e shard fails, to avoid overlapping their model load. +The `All integration tests` check requires every selected integration job to pass; full coverage +does not depend on a label or a manual request. -Run all live journeys with explicit agent versions through -`scripts/run_integration.py`. The separate integration workflow is not configured -in this checkout; collection and package checks do not count as live passes. +The existing e2e workflow runs seven parallel shards: gateway checks plus one for +each of Claude, Codex, Gemini, OpenCode, Copilot, and Pi. Each agent shard installs +its own CLI. Configure-subset checks run in the Claude shard because configuration +invokes the Claude CLI. The Claude shard also runs the existing tracing test file, whose +pre-existing skip remains in place. The `All agent tests` check requires every shard to pass. +Check names describe the coverage: `Unit tests`, `Gateway API tests`, +`Agent launch tests · Claude`, `Smoke journeys · Claude`, and +`Full journeys · Claude · Configure` (with the other agents/groups named likewise). +Unit tests still run as one job. Both matrices use `fail-fast: false` so one +failure does not cancel other coverage. + +The small `test` and `e2e` compatibility gates retain the exact status contexts +required by the repository's branch rules. `test` requires `Unit tests`; `e2e` +requires both `All agent tests` and the complete integration workflow. A failed +or skipped dependency fails the gate, and a running integration suite keeps it +pending. The descriptive jobs provide the actual coverage and diagnostics. ## Gaps and deferred scope @@ -75,7 +90,7 @@ in this checkout; collection and package checks do not count as live passes. | Provider switching, relayed/subscription MPS | Not covered by the four provider journeys | | TUI initial prompt supplied on the launch command line | Not yet covered; headless prompt arguments are covered | | Follow-up turns and conversation resume | Not covered; reopen proves startup, not conversation resume | -| Claude child model identity | Verified only when the actual start event reports it; missing fields remain unknown | +| Claude/Codex interactive smart routing | Deferred at the user's request; routing jobs and live journeys removed. Unit/component routing tests remain, but do not establish live routing behavior. | | Full allow/deny tool-permission matrix | Not covered; onboarding/trust uses actual TUI choices | | Desktop Codex app, Isaac itself, auto-upgrades | Not covered by command forwarding or pinned-version tests | | Native macOS/Windows managed settings, resize/signals | Separate platform coverage needed | diff --git a/tests/integration/README.md b/tests/integration/README.md index 3b9db5752..e3b5bf159 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -17,7 +17,7 @@ Pytest and the PTY/screen libraries (pexpect and pyte) live in a different virtu missing application dependency. No packages are installed into your existing agent installations or checkout's `.venv`. Native live runs refuse existing machine-wide Claude/Codex configuration, which -could override the selected workspace even with a fresh home. Use the container +could override the selected workspace even with a fresh home. Use a clean VM in that case; the runner never edits or bypasses those managed settings. Use the existing e2e workspace and its `DATABRICKS_BEARER` credential. Locally, @@ -43,7 +43,7 @@ for an npm mirror if public npm is unavailable. Older releases that only provide the `ucode` command require `--entry-point ucode`. Select one agent by providing only its version. Exact agent versions are -required; floating `latest`, caret, and tilde versions are rejected. Provider and default-routing CUJs use the workspace's configuration and need no model input. Cases that +required; floating `latest`, caret, and tilde versions are rejected. Hosted provider CUJs use the workspace's configuration and need no model input. Cases that exercise explicit model arguments use a real `system.ai` model already discovered by `ug configure`, recorded in that case's `model.json`. Optional `--claude-model` and `--codex-model` overrides reproduce a particular model-related failure. @@ -80,8 +80,6 @@ All user journeys are top-level tests. There is no separate regressions category ```text test_ug_configure_claude.py # Databricks Hosted and Anthropic MPS test_ug_configure_codex.py # Databricks Hosted and OpenAI MPS -test_smart_routing_claude.py # first prompt, subagent, explicit model -test_smart_routing_codex.py # first prompt, subagent, explicit model test_ug_claude_headless.py # script prompts, models, caller settings test_ug_codex_headless.py # script prompts and model arguments test_ug_claude_commands.py # command help forwarding @@ -102,43 +100,47 @@ process/terminal mechanics, evidence, and cleanup. Fixtures supply an isolated session and credentials; none manufacture or configure application state. Provider CUJs use normal ug validation, then require their own completed -interactive task. Routing CUJs skip the preliminary validation prompt and require -the actual routed TUI task instead. Tests disable optional Databricks AI Tools and +interactive task. Other journeys skip preliminary validation when they provide +their own task or command assertions. Tests disable optional Databricks AI Tools and pass `--skip-upgrade` to preserve the selected version. They use real onboarding -and trust choices, without seeded acceptance or disabled agent sandboxing. +and trust choices, without seeded acceptance or disabled agent sandboxing. If a +routed child asks to locate the random fixture beneath the disposable project, +the terminal driver accepts that exact read-only command through Claude's real +permission dialog; any broader permission request fails immediately. A fixture file contains an unpredictable value absent from the prompt. Success requires an assistant answer in the real agent transcript containing that value, -plus normal TUI exit. Codex evidence requires its task-complete event. Subagent -CUJs require a separate child transcript, child answer, and a correlated routing -decision. Claude uses its real child-start audit; its model is unknown when the -event omits it. Codex requires native parent linkage and a completed child turn -using the routed model, excluding inherited parent turns. +plus normal TUI exit. Codex evidence requires its task-complete event. +Interactive smart-routing journeys and their Claude/Codex CI shards are deferred +at the user's request. Unit/component routing tests remain; live first-prompt, +subagent routing, and interactive explicit-model bypass are not covered. MPS CUJs select the existing services already used by e2e: - Claude: `main.ucode.ci_e2e_anthropic_nonrelay_mps`. -- Codex: `main.ucode.ci_openai_mps`. +- Codex: `main.ucode.ci_openai_mps`, using its allowed `gpt-5-nano` model. Use `--claude-provider` / `--codex-provider` to reproduce another existing service. -Those names are recorded in `versions.json`. No service is created or modified. +Use `--codex-provider-model` when that OpenAI service allows a different model. +Those choices are recorded in `versions.json`. No service is created or modified. A missing service or permission fails the selected CUJ, rather than skipping it. -There are **45 live cases** (including 10 TUI journeys) and **3 installation +There are **39 live cases** (including 4 TUI journeys) and **3 installation checks** with both agents. See the named coverage and gaps matrix in [../README.md](../README.md). ```bash # Append one of these selections to the runner command: -- -m live # default: all live user journeys --- -m tui # ten complete interactive TUI journeys +-- -m smoke # four Hosted configure/TUI and headless argument journeys +-- -m tui # four complete provider-configuration TUI journeys -- -k test_ug_codex_app_server_client_initializes # one named journey and its variants # Use --installation-only before -- for package checks without credentials. ``` The old focused checks are now descriptive CUJs with setup and outcomes visible -in each test. Duplicate boot-only checks are incorporated into the Databricks, -first-prompt, and explicit-model TUI journeys. Real failures, including generated +in each test. Duplicate boot-only checks are incorporated into the Databricks +configuration TUI journeys. Real failures, including generated config left after revert and banners on app-server stdout, remain assertions. MCP/skills functionality, tracing, the broad configure-option matrix, and other agents are outside this focused revision. @@ -173,8 +175,145 @@ Selection after `--` accepts `-k`, `-m`, `-x`, and `--maxfail`; configuration an report paths cannot be overridden. `--installation-only` always restricts the selection to installation checks, including when additional filters are used. +## Run in GitHub Actions + +The **CI** workflow calls **Integration** on pull requests and pushes to `main`, +after its existing e2e shards finish (even if one fails). +It runs directly on fresh GitHub Ubuntu VMs, not inside the optional Docker image. +Local native runs use the same runner; Colima/Docker provides a separate Linux +container option. Matching dependency versions does not make those OS environments identical. +Its installation job needs no credentials. For same-repository PRs, the live jobs +reuse the existing `UCODE_TEST_WORKSPACE` and `DATABRICKS_BEARER` secrets. Fork PRs +run installation checks only because they cannot receive those secrets. + +The workspace check requires the secret to match +`https://eng-ml-inference-team-us-east-1.cloud.databricks.com` (a trailing slash +is accepted). It never changes the secret or switches workspaces. There is no CI +model-discovery or model-selection job. Real `ug configure` performs its normal +workspace discovery inside each test; only explicit-model scenarios choose and +record a discovered `system.ai` model as a test argument. +Every same-repository PR and push to `main` runs **Smoke journeys**, followed by +**Full journeys** even if smoke fails. Smoke runs the Hosted configure/TUI and headless +argument journey for each agent (four cases, two agent jobs). Full runs all 39 +live cases, including those smoke cases, in six disjoint shards: + +| Group, per agent | Marker | Keyword filter | +| --- | --- | --- | +| configure | `live and tui` | `configure` | +| headless | `live and not tui` | `headless` | +| commands | `live and not tui` | `not headless` | + +Each shard also selects `claude` or `codex` and installs only that CLI. The +commands group includes lifecycle and app-server journeys. Cases remain serial +inside each fresh VM because configure/revert can touch machine-level settings; +separate runners isolate those writes as well as the PTYs. The six full shards +run one at a time to avoid bursts against the shared workspace/model quota. +Both matrices use +`fail-fast: false` and upload uniquely named evidence even when another shard fails. +The **All integration tests** check requires installation, workspace validation, smoke, and +all full shards to pass. The existing required `e2e` context also waits for the +complete integration workflow, so integration cannot still be running when +that gate passes. Full coverage on PRs needs no label or opt-in. + +Each job uses fresh consumer dependency resolution. There is no default dependency +matrix. Manual dispatch accepts an +optional `dependency` such as `tomlkit==0.14.0`, equivalent to the local runner's +`--dependency` option. Jobs use Ubuntu 22.04; newer Ubuntu runner +policies prevented Codex's bubblewrap tool from reading even the test file in the +first run. The agent sandbox is not disabled or bypassed. +The workflow consumes the stored bearer; it does not mint or refresh credentials. + +For a manual run, use **Actions → Integration → Run workflow**, select the branch, +and choose `full` (default), `smoke`, `tui`, or `installation`. `live` remains an +alias for `full`. Manual subsets are explicit: `smoke` runs just the four smoke +cases; `tui` runs all four TUI cases across the configure shards. Installation +checks always run. Set the ug/agent versions. From the CLI: + +```bash +gh workflow run integration.yml -R databricks/unity-gateway --ref YOUR_BRANCH \ + -f suite=full -f ug_version=checkout \ + -f claude_version=2.1.268 -f codex_version=0.154.0 +gh run list -R databricks/unity-gateway --workflow integration.yml +gh run watch RUN_ID -R databricks/unity-gateway --exit-status +``` + +GitHub enables manual dispatch once the workflow exists on the default branch. +Before this PR merges, CI's pull-request event calls the workflow. Missing +credentials or a workspace mismatch fail the workspace job. Expired or invalid +credentials fail the actual workspace calls. Those failures do not count as live +test passes. + +## Reproduce and debug a CI failure locally + +Use the same runner and the failing job's artifacts. A new developer machine +needs Python 3.12+, uv, Node/npm, Databricks CLI, and its own authorized login for +the CI workspace. Select that local profile explicitly; CI secrets are not downloaded. + +```bash +gh run download RUN_ID -R databricks/unity-gateway \ + -n integration-full-claude-configure -D .integration-runs/from-ci +``` + +Use `integration-full-AGENT-GROUP` for a full shard, `integration-smoke-AGENT` for +smoke, or `integration-installation` for package failures. Older runs used +`integration-cujs` or numbered `integration-live-*` artifacts; download the name +shown on that run. Read `versions.json` for the +exact agent versions, model overrides, entry point, platform and source revision. +For an explicit-model case without a runner override, read its `model.json` for +the exact model used. Basic boot cases require no model arguments. Use +the archived wheel so a changed checkout cannot alter the reproduction: + +```bash +python3.12 scripts/run_integration.py \ + --ug-wheel .integration-runs/from-ci/wheels/EXACT_WHEEL.whl \ + --entry-point ug \ + --claude-version CLAUDE_VERSION_FROM_REPORT \ + --workspace https://eng-ml-inference-team-us-east-1.cloud.databricks.com \ + --profile YOUR_E2E_PROFILE \ + --constraints .integration-runs/from-ci/dependencies.txt \ + --npm-lock .integration-runs/from-ci/npm-lock.json \ + --output .integration-runs/repro-1 \ + -- -k test_ug_configure_claude_databricks +``` + +For an explicit-model failure, also pass the recorded `--claude-model` or +`--codex-model`. For a release run without an archived wheel, use the reported `--ug-version`. +The example replays a Claude shard; for Codex use only `--codex-version` and its +test filter. Match the report's selected agents, `pytest_args`, and suite revision +to replay an entire shard; a `-k` filter can reproduce one case independently. +Match Python and Node versions from the report too. `npm-lock.json` replay must +use the same OS/architecture as the original run; add `--platform linux/amd64` +to both `docker build` and `docker run` on an ARM Mac to match GitHub's Ubuntu runner. Changing platforms or +resolving a fresh npm lock is a new comparison, not an exact dependency replay. + +Use `-- -m tui` for the four interactive journeys or +`-- -k test_ug_configure_codex_databricks` to narrow a failure. Each rerun needs a new output directory. Inspect: + +- `junit.xml` for the failing case and assertion. +- `artifacts//command-*.json` for the real argv, exit status, stdout and stderr. +- `artifacts//first-session.json`, `provider-session.json`, or `reopen.json` + for rendered terminal screens, raw terminal + output, keyboard actions, exit status, routing logs, and actual agent-session records. +- `install.log` for resolution/bootstrap failures. + +Unknown onboarding screens fail with their actual screen text. Update terminal +selectors only after confirming the agent's intended UI changed; do not seed its +onboarding state or relax the prompt/task assertions. Test homes are deleted after +each case; redacted diagnostics remain. For manual interaction, configure a fresh +home with the same installed binaries and recorded public CLI arguments. +Definitive API errors and the client's exhausted retry limit fail the TUI wait +immediately with the actual screen. A transient 429/503 while the client is still +retrying is not treated as terminal; the suite adds no task retries of its own. + ## Colima / Docker +Docker is an optional installation/command-check environment, not a guaranteed +replacement for the native Linux live suite. Default Colima/Docker security +policies can reject Codex's user namespaces; the image also lacks `sudo` needed +by machine-level configure/revert journeys. Do not disable the agent sandbox or +use privileged/unconfined containers to turn these into passing results. Use a +clean Ubuntu 22.04 VM for the complete live run below. + Colima provides the Linux Docker engine on macOS. The optional image pins the Python, Node, uv, and Databricks toolchain; the same runner selects ug and agent versions inside it. Build from the repository root: @@ -194,7 +333,7 @@ docker run --rm --init \ ug-integration \ --ug-version YOUR_RELEASE_VERSION \ --claude-version 2.1.268 --codex-version 0.154.0 \ - -- -m live + --installation-only ``` Use a new results volume for each run, or pass a new `--output /results/NAME`. @@ -206,3 +345,63 @@ native runs also depend on the host's OS and toolchain. The explicit build archive includes only the runner and integration files, even with legacy Docker builders that ignore per-Dockerfile ignore rules. It also omits macOS extended attributes that Linux cannot unpack. + +### Complete local checkout run on the Databricks network + +Run from the checkout root in Bash on a clean Ubuntu 22.04 machine/VM with +Python 3.12, uv 0.9.8, Node 22.19.0/npm, Databricks CLI 1.9.0, `sudo`, and +`bubblewrap`. Install bubblewrap with `sudo apt-get install bubblewrap`. Verify +the normal sandbox before starting: + +```bash +bwrap --ro-bind / / --unshare-user --proc /proc --dev /dev /usr/bin/true +``` + +On macOS, Lima can supply a separate VM with no host mounts: + +```bash +limactl start --name ug-integration --plain --cpus 2 --memory 4 --disk 12 \ + --yes template:ubuntu-22.04 +limactl shell ug-integration +``` + +Install the prerequisites and copy/clone the checkout inside that VM. On recent +Apple Silicon, the original Ubuntu 22.04 ARM kernel can crash `cryptography` +with `Illegal instruction` before ug starts. Install Ubuntu's supported +`linux-generic-hwe-22.04` package and restart the VM. Do not work around it by +altering Python dependencies or weakening tests. Linux/ARM is not an exact +replay of GitHub's Linux/AMD64 environment; the run records its actual platform. + +Use your authorized login and explicitly selected profile; never download CI +secrets. The runner builds the checkout wheel and isolates the installed agents, +ug, test dependencies, and per-case homes: + +```bash +set -euo pipefail +integration_profile=eng-ml-inference-team-us-east-1 +integration_workspace=https://eng-ml-inference-team-us-east-1.cloud.databricks.com +integration_index=https://pypi-proxy.cloud.databricks.com/simple +integration_registry=https://npm-proxy.cloud.databricks.com/ + +# Keep this setting on both login and token retrieval when using plaintext storage. +export DATABRICKS_AUTH_STORAGE=plaintext +databricks auth login --host "$integration_workspace" --profile "$integration_profile" +export DATABRICKS_BEARER +DATABRICKS_BEARER=$(databricks auth token --host "$integration_workspace" \ + --profile "$integration_profile" --output json | jq -er '.access_token | select(length > 0)') +uv run --no-project --python 3.12 python scripts/run_integration.py \ + --python 3.12 --ug-version checkout --workspace "$integration_workspace" \ + --default-index "$integration_index" --npm-registry "$integration_registry" \ + --claude-version 2.1.268 --codex-version 0.154.0 -- -m live +unset DATABRICKS_BEARER +``` + +This runs all 39 live cases. For the three installation checks, run the same +runner/version/index arguments with `--installation-only` and omit `-- -m live`; +no bearer or workspace is needed. Results remain under `.integration-runs/`. +Each invocation needs a new output directory; an existing one is rejected. +The runner returns nonzero on installation or test failure. + +Do not drop the mirror flags if public registries resolve to `127.0.0.1` or return +`ECONNREFUSED`. The runner deliberately ignores host `.npmrc` and resolver settings. +Outside the Databricks network, use reachable package indexes explicitly instead. diff --git a/tests/integration/conftest.py b/tests/integration/conftest.py index f3ce72b29..37b67b78c 100644 --- a/tests/integration/conftest.py +++ b/tests/integration/conftest.py @@ -91,3 +91,8 @@ def claude_provider(): @pytest.fixture(scope="session") def codex_provider(): return os.environ["UG_INTEGRATION_CODEX_PROVIDER"] + + +@pytest.fixture(scope="session") +def codex_provider_model(): + return os.environ["UG_INTEGRATION_CODEX_PROVIDER_MODEL"] diff --git a/tests/integration/pytest.ini b/tests/integration/pytest.ini index ed06ccb80..a89db1660 100644 --- a/tests/integration/pytest.ini +++ b/tests/integration/pytest.ini @@ -4,6 +4,7 @@ addopts = --strict-markers --tb=short markers = installation: installed-package checks that require no workspace live: requires the real workspace used by the existing e2e suite + smoke: Databricks Hosted configure/TUI and headless prompt for each agent tui: real interactive terminal boot, keyboard input, exit and reopen claude: only runs when Claude Code is explicitly selected codex: only runs when Codex is explicitly selected diff --git a/tests/integration/test_smart_routing_claude.py b/tests/integration/test_smart_routing_claude.py deleted file mode 100644 index 1834760e1..000000000 --- a/tests/integration/test_smart_routing_claude.py +++ /dev/null @@ -1,106 +0,0 @@ -"""CUJs: real first-prompt and child-task routing in Claude's TUI.""" - -import pytest -from utils.evidence import FileTask, assert_subagent_routed -from utils.terminal import AgentTerminal - -pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.claude] - - -def test_smart_routing_claude_first_prompt(live_session, workspace): - """Scenario: enable smart routing and submit Claude's first interactive prompt. - - Expected: a real gateway decision selects a model, the prompt is replayed, - and the assistant completes the task before a normal exit. Boot alone fails. - """ - session = live_session - task = FileTask(session) - session.run( - "configure", - "--agents", - "claude", - "--workspaces", - workspace, - "--skip-validate", - "--skip-upgrade", - "--disable-databricks-ai-tools", - ) - - command = [str(session.binary), "claude", "--enable-smart-routing"] - with AgentTerminal(session, "claude", command, "first-prompt") as tui: - tui.boot() - assert "[ROUTE]" not in session.routing_log("claude"), "Boot must not route a prompt" - tui.submit(task.prompt) - tui.wait_for_task(task) - log = session.routing_log("claude") - assert "[ROUTE] first prompt ->" in log, log - assert "[REPLAY] first prompt submitted" in log, log - tui.exit_normally() - task.assert_completed(session, "claude") - with AgentTerminal(session, "claude", command, "reopen") as tui: - tui.boot() - tui.check_input_and_exit() - - -def test_smart_routing_claude_subagent(live_session, workspace): - """Scenario: ask Claude to delegate a file-reading task with routing enabled. - - Expected: a real child session has a correlated gateway routing decision, - the child returns the file value, and the parent returns that result. - Child model identity remains explicitly unknown if its event omits it. - """ - session = live_session - task = FileTask(session) - session.run( - "configure", - "--agents", - "claude", - "--workspaces", - workspace, - "--skip-validate", - "--skip-upgrade", - "--disable-databricks-ai-tools", - ) - - command = [str(session.binary), "claude", "--enable-smart-routing"] - with AgentTerminal(session, "claude", command, "subagent-task") as tui: - tui.boot() - tui.submit(task.delegate_prompt) - tui.wait_for_task(task, timeout=240) - task.assert_completed(session, "claude", child=True) - assert_subagent_routed(session, "claude", task) - tui.exit_normally() - task.assert_completed(session, "claude") - - -def test_smart_routing_claude_explicit_model_bypasses_routing(live_session, workspace): - """Scenario: request an explicit model while launching an interactive routing session. - - Expected: claude completes the real task with the caller's model choice, - starts no routing wrapper, exits normally, and reopens with usable input. - """ - session = live_session - task = FileTask(session) - session.run( - "configure", - "--agents", - "claude", - "--workspaces", - workspace, - "--skip-validate", - "--skip-upgrade", - "--disable-databricks-ai-tools", - ) - model = session.model_for_explicit_case("claude") - command = [str(session.binary), "claude", "--enable-smart-routing", "--model", model] - with AgentTerminal(session, "claude", command, "explicit-model") as tui: - tui.boot() - tui.submit(task.prompt) - tui.wait_for_task(task) - tui.exit_normally() - task.assert_completed(session, "claude") - session.assert_not_routed() - with AgentTerminal(session, "claude", command, "reopen") as tui: - tui.boot() - tui.check_input_and_exit() - session.assert_not_routed() diff --git a/tests/integration/test_smart_routing_codex.py b/tests/integration/test_smart_routing_codex.py deleted file mode 100644 index 64045d8e3..000000000 --- a/tests/integration/test_smart_routing_codex.py +++ /dev/null @@ -1,106 +0,0 @@ -"""CUJs: real first-prompt and child-task routing in Codex's TUI.""" - -import pytest -from utils.evidence import FileTask, assert_subagent_routed -from utils.terminal import AgentTerminal - -pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.codex] - - -def test_smart_routing_codex_first_prompt(live_session, workspace): - """Scenario: enable smart routing and submit Codex's first interactive prompt. - - Expected: a real gateway decision selects a model and Codex completes the - file-reading task before a normal exit. A routing fallback does not pass. - """ - session = live_session - task = FileTask(session) - session.run( - "configure", - "--agents", - "codex", - "--workspaces", - workspace, - "--skip-validate", - "--skip-upgrade", - "--disable-databricks-ai-tools", - ) - - command = [str(session.binary), "codex", "--enable-smart-routing"] - with AgentTerminal(session, "codex", command, "first-prompt") as tui: - tui.boot() - assert "[ROUTE]" not in session.routing_log("codex"), "Boot must not route a prompt" - tui.submit(task.prompt) - tui.wait_for_task(task) - log = session.routing_log("codex") - assert "[ROUTE] selected" in log, log - assert "selection failed" not in log and "settings/update timed out" not in log, log - tui.exit_normally() - task.assert_completed(session, "codex") - with AgentTerminal(session, "codex", command, "reopen") as tui: - tui.boot() - tui.check_input_and_exit() - - -def test_smart_routing_codex_subagent(live_session, workspace): - """Scenario: ask Codex to delegate a file-reading task with routing enabled. - - Expected: a real child session has a correlated gateway routing decision, - the child returns the file value, and the parent returns that result. - The native child turn must use the routed model; inherited parent turns do not count. - """ - session = live_session - task = FileTask(session) - session.run( - "configure", - "--agents", - "codex", - "--workspaces", - workspace, - "--skip-validate", - "--skip-upgrade", - "--disable-databricks-ai-tools", - ) - - command = [str(session.binary), "codex", "--enable-smart-routing"] - with AgentTerminal(session, "codex", command, "subagent-task") as tui: - tui.boot() - tui.submit(task.delegate_prompt) - tui.wait_for_task(task, timeout=240) - task.assert_completed(session, "codex", child=True) - assert_subagent_routed(session, "codex", task) - tui.exit_normally() - task.assert_completed(session, "codex") - - -def test_smart_routing_codex_explicit_model_bypasses_routing(live_session, workspace): - """Scenario: request an explicit model while launching an interactive routing session. - - Expected: codex completes the real task with the caller's model choice, - starts no routing wrapper, exits normally, and reopens with usable input. - """ - session = live_session - task = FileTask(session) - session.run( - "configure", - "--agents", - "codex", - "--workspaces", - workspace, - "--skip-validate", - "--skip-upgrade", - "--disable-databricks-ai-tools", - ) - model = session.model_for_explicit_case("codex") - command = [str(session.binary), "codex", "--enable-smart-routing", "--", "--model", model] - with AgentTerminal(session, "codex", command, "explicit-model") as tui: - tui.boot() - tui.submit(task.prompt) - tui.wait_for_task(task) - tui.exit_normally() - task.assert_completed(session, "codex") - session.assert_not_routed() - with AgentTerminal(session, "codex", command, "reopen") as tui: - tui.boot() - tui.check_input_and_exit() - session.assert_not_routed() diff --git a/tests/integration/test_ug_claude_headless.py b/tests/integration/test_ug_claude_headless.py index 0f537838a..0e0fde4dd 100644 --- a/tests/integration/test_ug_claude_headless.py +++ b/tests/integration/test_ug_claude_headless.py @@ -8,6 +8,7 @@ pytestmark = [pytest.mark.live, pytest.mark.claude] +@pytest.mark.smoke def test_ug_claude_headless_prompt_argument(live_session, workspace): """Scenario: configure claude and submit a headless prompt via argument. diff --git a/tests/integration/test_ug_codex_headless.py b/tests/integration/test_ug_codex_headless.py index 0b83f54d1..68f07c122 100644 --- a/tests/integration/test_ug_codex_headless.py +++ b/tests/integration/test_ug_codex_headless.py @@ -6,6 +6,7 @@ pytestmark = [pytest.mark.live, pytest.mark.codex] +@pytest.mark.smoke def test_ug_codex_headless_prompt_argument(live_session, workspace): """Scenario: configure codex and submit a headless prompt via argument. diff --git a/tests/integration/test_ug_configure_claude.py b/tests/integration/test_ug_configure_claude.py index 37fc5ac5c..d56fb90be 100644 --- a/tests/integration/test_ug_configure_claude.py +++ b/tests/integration/test_ug_configure_claude.py @@ -7,6 +7,7 @@ pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.claude] +@pytest.mark.smoke def test_ug_configure_claude_databricks(live_session, workspace): """Scenario: configure Claude with Databricks Hosted and use its TUI. diff --git a/tests/integration/test_ug_configure_codex.py b/tests/integration/test_ug_configure_codex.py index c2759c604..f7df9b4ed 100644 --- a/tests/integration/test_ug_configure_codex.py +++ b/tests/integration/test_ug_configure_codex.py @@ -7,6 +7,7 @@ pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.codex] +@pytest.mark.smoke def test_ug_configure_codex_databricks(live_session, workspace): """Scenario: configure Codex with Databricks Hosted and use its TUI. @@ -42,11 +43,14 @@ def test_ug_configure_codex_databricks(live_session, workspace): tui.check_input_and_exit() -def test_ug_configure_codex_openai_mps(live_session, workspace, codex_provider): +def test_ug_configure_codex_openai_mps( + live_session, workspace, codex_provider, codex_provider_model +): """Scenario: choose the real OpenAI MPS in ug configure's provider picker. Expected: ug saves that provider, and launching Codex without --provider - uses the saved choice to complete a file-reading task and exit normally. + uses the saved choice with one of its allowed models to complete a + file-reading task and exit normally. """ session = live_session task = FileTask(session) @@ -67,7 +71,8 @@ def test_ug_configure_codex_openai_mps(live_session, workspace, codex_provider): assert session.workspace_state()["provider_services"]["codex"] == codex_provider assert codex_provider in session.run("status").stdout - with AgentTerminal(session, "codex", [str(session.binary), "codex"], "provider-session") as tui: + command = [str(session.binary), "codex", "--", "--model", codex_provider_model] + with AgentTerminal(session, "codex", command, "provider-session") as tui: tui.boot() tui.submit(task.prompt) tui.wait_for_task(task, timeout=300) diff --git a/tests/integration/utils/evidence.py b/tests/integration/utils/evidence.py index 6cfafdac7..052b75442 100644 --- a/tests/integration/utils/evidence.py +++ b/tests/integration/utils/evidence.py @@ -1,10 +1,22 @@ """Read real agent transcripts and ug routing records without changing them.""" import json +import re import uuid from pathlib import Path +def assert_no_terminal_api_error(screen: str) -> None: + """Fail on definitive client errors, not an in-progress transient retry.""" + error = re.search( + r"unexpected status (?:400|401|403|404|405|409|422)\b|PERMISSION_DENIED" + r"|exceeded retry limit", + screen, + re.IGNORECASE, + ) + assert error is None, "Agent returned a terminal API error:\n" + screen + + def read_jsonl(path: Path) -> list[dict]: if not path.is_file(): return [] diff --git a/tests/integration/utils/terminal.py b/tests/integration/utils/terminal.py index ea6b05208..1a68b72cd 100644 --- a/tests/integration/utils/terminal.py +++ b/tests/integration/utils/terminal.py @@ -16,7 +16,7 @@ import pexpect import pyte -from .evidence import agent_sessions +from .evidence import agent_sessions, assert_no_terminal_api_error class TerminalScreen(pyte.Screen): @@ -282,8 +282,36 @@ def check_input_and_exit(self): self.exit_normally() def wait_for_task(self, task, timeout=180): + permission_in_progress = False + + def completed(screen): + nonlocal permission_in_progress + assert_no_terminal_api_error(screen) + if "Do you want to proceed?" in screen: + if permission_in_progress: + return False + # A routed Claude child may locate the fixture from the + # disposable project root before reading it. Interact with + # that real permission dialog, but never approve a broader or + # mutating command just because a model requested it. + project_root = re.escape(str(self.session.cwd.parent)) + filename = re.escape(task.filename) + safe_find = re.search( + rf'(?m)^\s*find {project_root} -name ["\']{filename}["\'] 2>/dev/null\s*$', + screen, + ) + first_yes = re.search(r"(?m)^\s*[›❯>]\s*1\.\s*Yes\s*$", screen) + assert self.agent == "claude" and safe_find and first_yes, ( + "Agent requested an unrecognized tool permission:\n" + screen + ) + self.send("\r", f"allow read-only search for {task.filename}") + permission_in_progress = True + return False + permission_in_progress = False + return task.completed(self.session, self.agent) + self.wait_for( - lambda text: task.completed(self.session, self.agent), + completed, "a completed assistant answer with the file's value", timeout=timeout, ) diff --git a/tests/test_e2e.py b/tests/test_e2e.py index a562a46b2..a7199ce23 100644 --- a/tests/test_e2e.py +++ b/tests/test_e2e.py @@ -498,7 +498,11 @@ def test_launch_codex_per_model(self, tmp_path, monkeypatch, e2e_state, e2e_work failures.append(f"model={model} timed out after {timeout_seconds}s") continue - if result.returncode != 0 or not (result.stdout or result.stderr).strip(): + if ( + result.returncode != 0 + or not result.stdout.strip() + or f"model: {codex.codex_model_id(model)}\n" not in result.stderr + ): # Keep a generous tail of stderr. codex-cli logs a non-fatal model-listing error # first and the actual cause last, so a short prefix reports the wrong problem — # at 200 chars the geography failure above read as a `/v1/models` routing error. diff --git a/tests/test_integration_contract.py b/tests/test_integration_contract.py index a637b9923..8268f2b61 100644 --- a/tests/test_integration_contract.py +++ b/tests/test_integration_contract.py @@ -4,6 +4,19 @@ from pathlib import Path +def _markers(nodes): + return { + node.attr + for root in nodes + for node in ast.walk(root) + if isinstance(node, ast.Attribute) + and isinstance(node.value, ast.Attribute) + and isinstance(node.value.value, ast.Name) + and node.value.value.id == "pytest" + and node.value.attr == "mark" + } + + def test_integration_suite_uses_only_public_process_boundaries(): violations = [] for path in (Path(__file__).parent / "integration").rglob("*.py"): @@ -39,6 +52,39 @@ def test_integration_suite_uses_only_public_process_boundaries(): assert not violations, "\n".join(violations) +def test_live_integration_cases_belong_to_exactly_one_ci_agent(): + for path in (Path(__file__).parent / "integration").glob("test_*.py"): + tree = ast.parse(path.read_text()) + module_marks = _markers( + node + for node in tree.body + if isinstance(node, ast.Assign) + and any( + isinstance(target, ast.Name) and target.id == "pytestmark" + for target in node.targets + ) + ) + for node in tree.body: + if isinstance(node, ast.FunctionDef) and node.name.startswith("test_"): + marks = module_marks | _markers(node.decorator_list) + if "live" in marks: + assert len(marks & {"claude", "codex"}) == 1, node.name + + +def test_smoke_covers_hosted_configuration_and_headless_for_both_agents(): + smoke = set() + for path in (Path(__file__).parent / "integration").glob("test_*.py"): + for node in ast.parse(path.read_text()).body: + if isinstance(node, ast.FunctionDef) and "smoke" in _markers(node.decorator_list): + smoke.add(node.name) + assert smoke == { + "test_ug_configure_claude_databricks", + "test_ug_configure_codex_databricks", + "test_ug_claude_headless_prompt_argument", + "test_ug_codex_headless_prompt_argument", + } + + def test_integration_tests_describe_the_scenario_and_expected_result(): root = Path(__file__).parent / "integration" violations = [] diff --git a/tests/test_integration_evidence.py b/tests/test_integration_evidence.py new file mode 100644 index 000000000..e057a2062 --- /dev/null +++ b/tests/test_integration_evidence.py @@ -0,0 +1,31 @@ +"""Unit checks for interpreting terminal evidence, not live agent substitutes.""" + +import pytest + +from tests.integration.utils.evidence import assert_no_terminal_api_error + + +@pytest.mark.parametrize( + "screen", + [ + "■ exceeded retry limit, last status: 429 Too Many Requests", + "■ unexpected status 403 Forbidden: PERMISSION_DENIED", + "■ unexpected status 401 Unauthorized", + ], +) +def test_terminal_api_failure_reports_the_actual_error(screen): + with pytest.raises(AssertionError, match="Agent returned a terminal API error") as error: + assert_no_terminal_api_error(screen) + assert screen in str(error.value) + + +@pytest.mark.parametrize( + "screen", + [ + "Reconnecting... 1/5 (unexpected status 429 Too Many Requests)", + "Reconnecting... 1/5 (unexpected status 503 Service Unavailable)", + "Working (5s · esc to interrupt)", + ], +) +def test_transient_retries_and_running_tasks_are_not_terminal_errors(screen): + assert_no_terminal_api_error(screen)