From 5b451e587528a2d24ea0f5557eadfc6d53453318 Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Mon, 14 Sep 2026 22:35:17 -0400 Subject: [PATCH 1/2] Run full integration in parallel agent lanes --- .github/workflows/integration.yml | 17 ++++++-------- tests/README.md | 8 ++++--- tests/integration/README.md | 39 +++++++++++++++++-------------- 3 files changed, 33 insertions(+), 31 deletions(-) diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index 604db3764..03c660bb6 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -123,7 +123,6 @@ jobs: DEPENDENCY: ${{ inputs.dependency }} AGENT: ${{ matrix.agent }} TEST_MARKER: live and smoke and ${{ matrix.agent }} - TEST_KEYWORD: '' ARTIFACT_NAME: integration-smoke-${{ matrix.agent }} steps: &live-steps - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 @@ -149,7 +148,7 @@ jobs: uv run --no-project --python 3.12 python scripts/run_integration.py \ --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ --default-index "$PACKAGE_INDEX" --output "$RUNNER_TEMP/ug-integration" \ - "${args[@]}" -- -m "$TEST_MARKER" -k "$TEST_KEYWORD" + "${args[@]}" -- -m "$TEST_MARKER" - name: Upload live test evidence if: ${{ !cancelled() }} uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 @@ -165,7 +164,7 @@ jobs: ${{ runner.temp }}/ug-integration/wheels/ full: - name: Full journeys · ${{ matrix.agent == 'claude' && 'Claude' || 'Codex' }} · ${{ matrix.group == 'configure' && 'Configure' || matrix.group == 'headless' && 'Headless' || 'Commands' }} + name: Full journeys · ${{ matrix.agent == 'claude' && 'Claude' || 'Codex' }} if: ${{ !cancelled() && inputs.suite != 'smoke' && needs.workspace.result == 'success' }} needs: [workspace, smoke] runs-on: ubuntu-22.04 @@ -174,20 +173,18 @@ jobs: timeout-minutes: 70 strategy: fail-fast: false - # All shards share one workspace/model quota. Serial execution retains - # every case without a burst of overlapping model requests. - max-parallel: 1 + # One VM per agent: reuse its install and run its cases serially. Allow + # Claude and Codex to overlap, but never launch concurrent same-agent shards. + max-parallel: 2 matrix: agent: [claude, codex] - group: ${{ fromJSON(inputs.suite == 'tui' && '["configure"]' || '["configure", "headless", "commands"]') }} env: UCODE_TEST_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} DEPENDENCY: ${{ inputs.dependency }} AGENT: ${{ matrix.agent }} - TEST_MARKER: ${{ matrix.group == 'configure' && 'live and tui' || 'live and not tui' }} and ${{ matrix.agent }} - TEST_KEYWORD: ${{ matrix.group == 'configure' && 'configure' || matrix.group == 'headless' && 'headless' || 'not headless' }} - ARTIFACT_NAME: integration-full-${{ matrix.agent }}-${{ matrix.group }} + TEST_MARKER: ${{ inputs.suite == 'tui' && 'live and tui' || 'live' }} and ${{ matrix.agent }} + ARTIFACT_NAME: integration-full-${{ matrix.agent }} steps: *live-steps cujs: diff --git a/tests/README.md b/tests/README.md index b7645b27a..ab148d001 100644 --- a/tests/README.md +++ b/tests/README.md @@ -58,8 +58,10 @@ dependency graph to reproduce a user's combination. Every relevant same-reposito PR and push to `main` runs both smoke and the full CUJ suite. Smoke covers the Databricks Hosted configure/TUI and headless argument journeys for both agents, in two parallel jobs. After smoke finishes, the full suite runs all 39 cases -across six serial jobs: Claude/Codex × configure, headless, and other -commands/lifecycle checks. CI calls integration after the existing e2e shards +across two parallel agent jobs: one Claude VM and one Codex VM, each running its +configure, headless, and commands/lifecycle cases serially. Each agent is installed +once for the full suite, and no two full jobs for the same agent overlap within a run. +CI calls integration after the existing e2e shards finish, including when an e2e shard fails, to avoid overlapping their model load. The `All integration tests` check requires every selected integration job to pass; full coverage does not depend on a label or a manual request. @@ -71,7 +73,7 @@ invokes the Claude CLI. The Claude shard also runs the existing tracing test fil pre-existing skip remains in place. The `All agent tests` check requires every shard to pass. Check names describe the coverage: `Unit tests`, `Gateway API tests`, `Agent launch tests · Claude`, `Smoke journeys · Claude`, and -`Full journeys · Claude · Configure` (with the other agents/groups named likewise). +`Full journeys · Claude` (with the other agents named likewise). Unit tests still run as one job. Both matrices use `fail-fast: false` so one failure does not cancel other coverage. diff --git a/tests/integration/README.md b/tests/integration/README.md index e3b5bf159..48d162002 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -195,23 +195,25 @@ record a discovered `system.ai` model as a test argument. Every same-repository PR and push to `main` runs **Smoke journeys**, followed by **Full journeys** even if smoke fails. Smoke runs the Hosted configure/TUI and headless argument journey for each agent (four cases, two agent jobs). Full runs all 39 -live cases, including those smoke cases, in six disjoint shards: +live cases, including those smoke cases, in two disjoint agent lanes: -| Group, per agent | Marker | Keyword filter | +| Agent lane | Marker | Cases | | --- | --- | --- | -| configure | `live and tui` | `configure` | -| headless | `live and not tui` | `headless` | -| commands | `live and not tui` | `not headless` | +| Claude | `live and claude` | 15 | +| Codex | `live and codex` | 24 | -Each shard also selects `claude` or `codex` and installs only that CLI. The -commands group includes lifecycle and app-server journeys. Cases remain serial +Each lane installs only its agent CLI, once, and runs all its configure, headless, +commands, lifecycle, and applicable app-server journeys. Cases remain serial inside each fresh VM because configure/revert can touch machine-level settings; -separate runners isolate those writes as well as the PTYs. The six full shards -run one at a time to avoid bursts against the shared workspace/model quota. -Both matrices use -`fail-fast: false` and upload uniquely named evidence even when another shard fails. +separate runners isolate those writes as well as the PTYs. Claude and Codex run +in parallel, with at most one full-suite job per agent in a workflow run. This +avoids six serial job startups without overlapping same-agent shards. The two +lanes still share workspace capacity, including with other PRs; this limit does +not guarantee freedom from rate limits. No test retries or assertion changes +compensate for capacity failures. Both matrices use `fail-fast: false` and upload +uniquely named evidence even when the other agent fails. The **All integration tests** check requires installation, workspace validation, smoke, and -all full shards to pass. The existing required `e2e` context also waits for the +both full lanes to pass. The existing required `e2e` context also waits for the complete integration workflow, so integration cannot still be running when that gate passes. Full coverage on PRs needs no label or opt-in. @@ -226,7 +228,7 @@ The workflow consumes the stored bearer; it does not mint or refresh credentials For a manual run, use **Actions → Integration → Run workflow**, select the branch, and choose `full` (default), `smoke`, `tui`, or `installation`. `live` remains an alias for `full`. Manual subsets are explicit: `smoke` runs just the four smoke -cases; `tui` runs all four TUI cases across the configure shards. Installation +cases; `tui` adds `and tui` to each agent lane's marker and runs all four TUI cases. Installation checks always run. Set the ug/agent versions. From the CLI: ```bash @@ -251,12 +253,13 @@ the CI workspace. Select that local profile explicitly; CI secrets are not downl ```bash gh run download RUN_ID -R databricks/unity-gateway \ - -n integration-full-claude-configure -D .integration-runs/from-ci + -n integration-full-claude -D .integration-runs/from-ci ``` -Use `integration-full-AGENT-GROUP` for a full shard, `integration-smoke-AGENT` for +Use `integration-full-AGENT` for a full lane, `integration-smoke-AGENT` for smoke, or `integration-installation` for package failures. Older runs used -`integration-cujs` or numbered `integration-live-*` artifacts; download the name +`integration-full-AGENT-GROUP`, `integration-cujs`, or numbered `integration-live-*` +artifacts; download the name shown on that run. Read `versions.json` for the exact agent versions, model overrides, entry point, platform and source revision. For an explicit-model case without a runner override, read its `model.json` for @@ -278,9 +281,9 @@ python3.12 scripts/run_integration.py \ For an explicit-model failure, also pass the recorded `--claude-model` or `--codex-model`. For a release run without an archived wheel, use the reported `--ug-version`. -The example replays a Claude shard; for Codex use only `--codex-version` and its +The example replays a Claude case; for Codex use only `--codex-version` and its test filter. Match the report's selected agents, `pytest_args`, and suite revision -to replay an entire shard; a `-k` filter can reproduce one case independently. +to replay an entire lane; a `-k` filter can reproduce one case independently. Match Python and Node versions from the report too. `npm-lock.json` replay must use the same OS/architecture as the original run; add `--platform linux/amd64` to both `docker build` and `docker run` on an ARM Mac to match GitHub's Ubuntu runner. Changing platforms or From 6c872167535b57d804b23fff6f2105ec961b6a44 Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Mon, 14 Sep 2026 22:46:59 -0400 Subject: [PATCH 2/2] Start integration alongside agent e2e tests --- .github/workflows/ci.yml | 5 ++--- tests/README.md | 5 +++-- tests/integration/README.md | 9 ++++++--- 3 files changed, 11 insertions(+), 8 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 4a2710734..47b48064a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -119,10 +119,9 @@ jobs: integration: name: Integration - # These suites use the same live workspace. Let agent launch coverage - # finish before starting integration requests, even if an agent failed. + # Start alongside unit/agent tests. Only the final required-e2e gate + # waits for both suites; an agent failure must not skip integration. if: ${{ !cancelled() }} - needs: e2e uses: ./.github/workflows/integration.yml secrets: inherit diff --git a/tests/README.md b/tests/README.md index ab148d001..6f86a9949 100644 --- a/tests/README.md +++ b/tests/README.md @@ -61,8 +61,9 @@ in two parallel jobs. After smoke finishes, the full suite runs all 39 cases across two parallel agent jobs: one Claude VM and one Codex VM, each running its configure, headless, and commands/lifecycle cases serially. Each agent is installed once for the full suite, and no two full jobs for the same agent overlap within a run. -CI calls integration after the existing e2e shards -finish, including when an e2e shard fails, to avoid overlapping their model load. +CI starts integration alongside unit tests and the existing e2e shards. Integration +does not wait for agent e2e or get skipped when an agent shard fails. These suites +share workspace capacity; overlapping their requests can still encounter rate limits. The `All integration tests` check requires every selected integration job to pass; full coverage does not depend on a label or a manual request. diff --git a/tests/integration/README.md b/tests/integration/README.md index 48d162002..ea8254922 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -178,7 +178,9 @@ selection to installation checks, including when additional filters are used. ## Run in GitHub Actions The **CI** workflow calls **Integration** on pull requests and pushes to `main`, -after its existing e2e shards finish (even if one fails). +starting alongside unit tests and the existing agent e2e shards. Integration has +no dependency on agent e2e; a failure there does not prevent integration from running. +The final required `e2e` check waits for both suites and requires both to succeed. It runs directly on fresh GitHub Ubuntu VMs, not inside the optional Docker image. Local native runs use the same runner; Colima/Docker provides a separate Linux container option. Matching dependency versions does not make those OS environments identical. @@ -208,8 +210,9 @@ inside each fresh VM because configure/revert can touch machine-level settings; separate runners isolate those writes as well as the PTYs. Claude and Codex run in parallel, with at most one full-suite job per agent in a workflow run. This avoids six serial job startups without overlapping same-agent shards. The two -lanes still share workspace capacity, including with other PRs; this limit does -not guarantee freedom from rate limits. No test retries or assertion changes +lanes still share workspace capacity with the concurrently running agent e2e +shards and other PRs; this limit does not guarantee freedom from rate limits. +No test retries or assertion changes compensate for capacity failures. Both matrices use `fail-fast: false` and upload uniquely named evidence even when the other agent fails. The **All integration tests** check requires installation, workspace validation, smoke, and