From cb4c585f8310dcb863d518462abfa059ace39e5c Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Fri, 11 Sep 2026 15:41:37 -0400 Subject: [PATCH 1/6] Add installed-agent integration and TUI boot tests with CI --- .github/workflows/integration.yml | 159 +++++++ .gitignore | 1 + README.md | 8 +- scripts/run_integration.py | 525 ++++++++++++++++++++++ tests/AGENTS.md | 90 ++++ tests/CLAUDE.md | 12 + tests/README.md | 69 +++ tests/conftest.py | 4 + tests/integration/Dockerfile | 26 ++ tests/integration/Dockerfile.dockerignore | 8 + tests/integration/README.md | 274 +++++++++++ tests/integration/conftest.py | 90 ++++ tests/integration/harness.py | 262 +++++++++++ tests/integration/pytest.ini | 9 + tests/integration/terminal.py | 192 ++++++++ tests/integration/test_installation.py | 23 + tests/integration/test_lifecycle.py | 58 +++ tests/integration/test_passthrough.py | 60 +++ tests/integration/test_tasks.py | 104 +++++ tests/integration/test_tui.py | 33 ++ tests/test_integration_contract.py | 39 ++ 21 files changed, 2045 insertions(+), 1 deletion(-) create mode 100644 .github/workflows/integration.yml create mode 100644 scripts/run_integration.py create mode 100644 tests/AGENTS.md create mode 100644 tests/CLAUDE.md create mode 100644 tests/README.md create mode 100644 tests/integration/Dockerfile create mode 100644 tests/integration/Dockerfile.dockerignore create mode 100644 tests/integration/README.md create mode 100644 tests/integration/conftest.py create mode 100644 tests/integration/harness.py create mode 100644 tests/integration/pytest.ini create mode 100644 tests/integration/terminal.py create mode 100644 tests/integration/test_installation.py create mode 100644 tests/integration/test_lifecycle.py create mode 100644 tests/integration/test_passthrough.py create mode 100644 tests/integration/test_tasks.py create mode 100644 tests/integration/test_tui.py create mode 100644 tests/test_integration_contract.py diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml new file mode 100644 index 000000000..1ec7e61c5 --- /dev/null +++ b/.github/workflows/integration.yml @@ -0,0 +1,159 @@ +name: Integration + +on: + pull_request: + paths: ['src/**', 'tests/**', 'scripts/run_integration.py', 'pyproject.toml', 'uv.lock', '.github/workflows/integration.yml'] + push: + branches: [main] + paths: ['src/**', 'tests/**', 'scripts/run_integration.py', 'pyproject.toml', 'uv.lock', '.github/workflows/integration.yml'] + workflow_dispatch: + inputs: + suite: + description: Which integration checks to run + type: choice + options: [all, tui, installation] + default: all + ug_version: + description: Exact ucode release, or checkout + default: checkout + required: true + entry_point: + description: Console command (older releases may only have ucode) + type: choice + options: [ug, ucode] + default: ug + claude_version: + description: Exact Claude Code version + default: 2.1.268 + required: true + codex_version: + description: Exact Codex version + default: 0.154.0 + required: true + index_url: + description: Python package index containing the requested ug version + default: https://pypi.org/simple + required: true + +permissions: + contents: read + +concurrency: + group: integration-${{ github.event.pull_request.number || github.ref }}-${{ github.event_name }} + cancel-in-progress: true + +env: + EXPECTED_WORKSPACE: https://eng-ml-inference-team-us-east-1.cloud.databricks.com + UG_VERSION: ${{ inputs.ug_version || 'checkout' }} + ENTRY_POINT: ${{ inputs.entry_point || 'ug' }} + CLAUDE_VERSION: ${{ inputs.claude_version || '2.1.268' }} + CODEX_VERSION: ${{ inputs.codex_version || '0.154.0' }} + PACKAGE_INDEX: ${{ inputs.index_url || 'https://pypi.org/simple' }} + +jobs: + installation: + name: Installation + runs-on: ubuntu-latest + timeout-minutes: 20 + steps: + - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + with: + fetch-depth: 0 + - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 + with: + node-version: 22.19.0 + - uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0 + with: + version: 0.9.8 + - name: Test a fresh installed package without credentials + run: | + python3 scripts/run_integration.py \ + --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ + --claude-version "$CLAUDE_VERSION" --codex-version "$CODEX_VERSION" \ + --index-url "$PACKAGE_INDEX" --installation-only --output "$RUNNER_TEMP/ug-integration" + - name: Upload installation evidence + if: ${{ !cancelled() }} + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 + with: + name: integration-installation + include-hidden-files: true + path: | + ${{ runner.temp }}/ug-integration/*.json + ${{ runner.temp }}/ug-integration/*.txt + ${{ runner.temp }}/ug-integration/*.xml + ${{ runner.temp }}/ug-integration/*.log + ${{ runner.temp }}/ug-integration/artifacts/ + ${{ runner.temp }}/ug-integration/wheels/ + + workspace: + name: Workspace + if: ${{ inputs.suite != 'installation' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) }} + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - name: Verify the existing e2e workspace and credential are configured + env: + UCODE_TEST_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} + DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} + run: | + python3 - <<'PY' + import os + expected = os.environ["EXPECTED_WORKSPACE"].rstrip("/") + actual = os.environ["UCODE_TEST_WORKSPACE"].strip().rstrip("/") + if actual != expected: + raise SystemExit("UCODE_TEST_WORKSPACE does not match the expected CI workspace.") + if not os.environ["DATABRICKS_BEARER"].strip(): + raise SystemExit("The existing e2e DATABRICKS_BEARER secret is missing.") + print("CI workspace matches; the existing e2e credential is present.") + PY + + live: + name: Live (${{ matrix.dependency }}) + needs: workspace + runs-on: ubuntu-latest + timeout-minutes: 45 + strategy: + fail-fast: false + matrix: + dependency: [unconstrained, tomlkit==0.14.0, tomlkit==0.15.1] + env: + UCODE_TEST_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} + DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} + DEPENDENCY: ${{ matrix.dependency }} + TEST_MARKER: ${{ inputs.suite == 'tui' && 'tui' || 'live' }} + steps: + - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + with: + fetch-depth: 0 + - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 + with: + node-version: 22.19.0 + - uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0 + with: + version: 0.9.8 + - uses: databricks/setup-cli@bdb89f81c11a5bd647fd55b585b7c396ec68a25a # v1.0.0 + - name: Run the selected live cases against the e2e workspace + shell: bash + run: | + args=() + if [[ "$DEPENDENCY" != unconstrained ]]; then + args+=(--dependency "$DEPENDENCY") + fi + python3 scripts/run_integration.py \ + --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ + --claude-version "$CLAUDE_VERSION" --codex-version "$CODEX_VERSION" \ + --index-url "$PACKAGE_INDEX" --output "$RUNNER_TEMP/ug-integration" \ + "${args[@]}" -- -m "$TEST_MARKER" + - name: Upload live test evidence + if: ${{ !cancelled() }} + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 + with: + name: integration-live-${{ strategy.job-index }} + include-hidden-files: true + path: | + ${{ runner.temp }}/ug-integration/*.json + ${{ runner.temp }}/ug-integration/*.txt + ${{ runner.temp }}/ug-integration/*.xml + ${{ runner.temp }}/ug-integration/*.log + ${{ runner.temp }}/ug-integration/artifacts/ + ${{ runner.temp }}/ug-integration/wheels/ diff --git a/.gitignore b/.gitignore index 007899c40..c616aacf8 100644 --- a/.gitignore +++ b/.gitignore @@ -6,3 +6,4 @@ dist/ .venv/ .DS_Store .isaac/ +.integration-runs/ diff --git a/README.md b/README.md index aad048d83..8a8a853fa 100644 --- a/README.md +++ b/README.md @@ -489,7 +489,13 @@ uv sync uv run ruff check . # lint ``` -4. For end-to-end testing against a real workspace: +4. For **integration tests** of installed ug and agent versions against the same + real workspace, see [the integration suite](tests/integration/README.md). + It uses separate processes and fresh homes, with no application mocks or + monkeypatching. The runner accepts ug/Claude/Codex versions and dependency + constraints to reproduce user issues. + + The existing e2e tests remain available separately: ```bash UCODE_TEST_WORKSPACE= uv run pytest tests/test_e2e.py -v diff --git a/scripts/run_integration.py b/scripts/run_integration.py new file mode 100644 index 000000000..f8cdbcf86 --- /dev/null +++ b/scripts/run_integration.py @@ -0,0 +1,525 @@ +"""Install a reproducible ug/agent combination and run black-box integration tests. + +Uses fresh virtualenvs and an isolated npm prefix, never the checkout's uv.lock +or the developer's installed agents. Only the live workspace is shared with e2e. +""" + +from __future__ import annotations + +import argparse +import contextlib +import datetime as dt +import hashlib +import json +import os +import platform +import re +import shutil +import signal +import subprocess +import sys +import xml.etree.ElementTree as ET +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +AGENT_PACKAGES = {"claude": "@anthropic-ai/claude-code", "codex": "@openai/codex"} + + +@contextlib.contextmanager +def managed_process(command, *, interrupt=False, **kwargs): + """Bound child lifetimes, including descendants that outlive their parent.""" + proc = subprocess.Popen(command, start_new_session=True, **kwargs) + try: + yield proc + finally: + # Give pytest a KeyboardInterrupt so its fixtures can clean up the + # separate process groups used by agent commands before pytest exits. + first_signal = signal.SIGINT if interrupt else signal.SIGTERM + with contextlib.suppress(ProcessLookupError): + os.killpg(proc.pid, first_signal) + try: + proc.wait(timeout=15 if interrupt else 5) + except subprocess.TimeoutExpired: + pass + with contextlib.suppress(ProcessLookupError): + os.killpg(proc.pid, signal.SIGKILL) + proc.wait(timeout=5) + + +def exact_npm_version(value: str) -> str: + if not re.fullmatch(r"\d+\.\d+\.\d+(?:-[0-9A-Za-z.-]+)?", value): + raise argparse.ArgumentTypeError( + "Use an exact version, for example 2.1.268; not latest/^/~." + ) + return value + + +def arguments(): + parser = argparse.ArgumentParser(description=__doc__) + source = parser.add_mutually_exclusive_group() + source.add_argument( + "--ug-version", default="checkout", help="Exact ucode release, or checkout." + ) + source.add_argument( + "--ug-wheel", type=Path, help="Previously built wheel to reproduce a release." + ) + parser.add_argument("--entry-point", choices=["ug", "ucode"], default="ug") + parser.add_argument("--claude-version", type=exact_npm_version) + parser.add_argument("--codex-version", type=exact_npm_version) + parser.add_argument("--claude-model", default=os.environ.get("UG_INTEGRATION_CLAUDE_MODEL")) + parser.add_argument("--codex-model", default=os.environ.get("UG_INTEGRATION_CODEX_MODEL")) + parser.add_argument("--python", default=sys.executable, help="Python 3.12+ path or uv version.") + parser.add_argument("--dependency", action="append", default=[], metavar="PACKAGE==VERSION") + parser.add_argument("--constraints", type=Path, help="Replay a previous dependencies.txt.") + parser.add_argument( + "--npm-lock", type=Path, help="Replay a previous npm-lock.json with npm ci." + ) + parser.add_argument( + "--index-url", default=os.environ.get("UV_INDEX_URL", "https://pypi.org/simple") + ) + parser.add_argument("--npm-registry", default="https://registry.npmjs.org") + parser.add_argument("--profile", help="Explicit Databricks profile to mint the live bearer.") + parser.add_argument("--workspace", default=os.environ.get("UCODE_TEST_WORKSPACE")) + parser.add_argument("--output", type=Path, help="New results directory; never reused.") + parser.add_argument("--installation-only", action="store_true", help="No workspace calls.") + parser.add_argument( + "pytest_args", nargs=argparse.REMAINDER, help="After --, pass pytest filters." + ) + args = parser.parse_args() + # Only selection/early-stop controls are accepted. Pytest configuration, + # plugins and report destinations are part of the suite's isolation contract. + filters = argparse.ArgumentParser(add_help=False) + filters.add_argument("-k") + filters.add_argument("-m") + filters.add_argument("-x", action="store_true") + filters.add_argument("--maxfail", type=int) + extra = args.pytest_args[1:] if args.pytest_args[:1] == ["--"] else args.pytest_args + selected = filters.parse_args(extra) + marker = selected.m + if args.installation_only: + marker = f"installation and ({marker})" if marker else "installation" + args.pytest_args = [] + for flag, value in (("-k", selected.k), ("-m", marker), ("--maxfail", selected.maxfail)): + if value is not None: + args.pytest_args.extend([flag, str(value)]) + if selected.x: + args.pytest_args.append("-x") + if not (args.claude_version or args.codex_version): + parser.error("Select --claude-version and/or --codex-version explicitly.") + if args.ug_version != "checkout" and not re.fullmatch( + r"[0-9][0-9A-Za-z.!+_-]*", args.ug_version + ): + parser.error( + "--ug-version must be an exact release, or checkout; use --ug-wheel for a file." + ) + for dependency in args.dependency: + if not re.fullmatch(r"[A-Za-z0-9_.-]+==[A-Za-z0-9_.!+-]+", dependency): + parser.error("--dependency requires an exact PACKAGE==VERSION constraint.") + if not args.installation_only: + if not args.workspace or not args.workspace.startswith("https://"): + parser.error( + "Set UCODE_TEST_WORKSPACE to the existing e2e workspace, or use --workspace." + ) + if not (args.profile or os.environ.get("DATABRICKS_BEARER", "").strip()): + parser.error("Provide the e2e DATABRICKS_BEARER or select --profile explicitly.") + return args + + +def main() -> int: + args = arguments() + if os.name != "posix": + raise SystemExit("This runner supports Linux and macOS. Use the container on other hosts.") + + def terminate(signum, frame): + raise KeyboardInterrupt + + signal.signal(signal.SIGTERM, terminate) + binaries = {name: shutil.which(name) for name in ("uv", "npm", "node", "databricks")} + required = ["uv", "npm", "node"] + ([] if args.installation_only else ["databricks"]) + missing = [name for name in required if not binaries[name]] + if missing: + raise SystemExit("Install these prerequisites first: " + ", ".join(missing)) + if not args.installation_only: + managed_paths = [] + if args.codex_version: + managed_paths.extend( + [Path("/etc/codex/managed_config.toml"), Path("/etc/codex/requirements.toml")] + ) + if args.claude_version: + managed_paths.append( + Path( + "/Library/Application Support/ClaudeCode/managed-settings.json" + if sys.platform == "darwin" + else "/etc/claude-code/managed-settings.json" + ) + ) + present = [str(path) for path in managed_paths if path.exists()] + if present: + raise SystemExit( + "Machine-wide agent settings can override the selected test workspace. " + "Use the integration container instead of this host: " + ", ".join(present) + ) + + stamp = dt.datetime.now(dt.UTC).strftime("%Y%m%dT%H%M%S.%fZ") + output = (args.output or ROOT / ".integration-runs" / stamp).resolve() + output.mkdir(parents=True, exist_ok=False) + build_log = output / "install.log" + # Do not inherit project environments, resolver constraints, pytest options, + # Python optimization, agent credentials, or npm settings from the caller. + keep = ( + "PATH", + "LANG", + "LC_ALL", + "SSL_CERT_FILE", + "SSL_CERT_DIR", + "REQUESTS_CA_BUNDLE", + "NODE_EXTRA_CA_CERTS", + "HTTPS_PROXY", + "HTTP_PROXY", + "NO_PROXY", + ) + base_env = {key: os.environ[key] for key in keep if key in os.environ} + build_home = output / "build-home" + build_home.mkdir() + base_env["HOME"] = str(build_home) + base_env["USERPROFILE"] = str(build_home) + base_env["npm_config_cache"] = str(output / "npm-cache") + base_env["npm_config_fetch_retries"] = "1" + base_env["npm_config_fetch_timeout"] = "30000" + base_env["UV_CACHE_DIR"] = str(output / "cache") + base_env["UV_INDEX_URL"] = args.index_url + bearer = os.environ.get("DATABRICKS_BEARER", "").strip() + + def redact(value: str) -> str: + return value.replace(bearer, "") if bearer else value + + def run(command, *, cwd=output, env=base_env, timeout=600) -> str: + timed_out = False + with managed_process( + [str(x) for x in command], + cwd=cwd, + env=env, + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + ) as proc: + try: + stdout, stderr = proc.communicate(timeout=timeout) + except subprocess.TimeoutExpired: + timed_out = True + if timed_out: + stdout, stderr = proc.communicate(timeout=5) + with build_log.open("a") as log: + log.write(redact(stdout + stderr)) + if timed_out: + raise RuntimeError(f"{command[0]} exceeded {timeout}s; see {build_log}") + if proc.returncode: + raise RuntimeError(f"{command[0]} failed; see {build_log}\n" + redact(stderr[-2000:])) + return stdout.strip() + + report = { + "requested": { + "ug": str(args.ug_wheel) if args.ug_wheel else args.ug_version, + "entry_point": args.entry_point, + "claude": args.claude_version, + "codex": args.codex_version, + "claude_model": args.claude_model, + "codex_model": args.codex_model, + "dependencies": args.dependency, + "workspace": args.workspace, + }, + "platform": platform.platform(), + "installation_only": args.installation_only, + } + manifest = output / "versions.json" + exitcode = 1 + try: + print(f"Installing selected versions. Results: {output}", flush=True) + uv = binaries["uv"] + runtime, testenv = output / "ug-runtime", output / "test-runtime" + for path in (runtime, testenv): + run([uv, "venv", "--python", args.python, path]) + python = runtime / "bin/python" + report["python"] = run([python, "--version"]) + report["uv"] = run([uv, "--version"]) + report["node"] = run([binaries["node"], "--version"]) + report["npm"] = run([binaries["npm"], "--version"]) + if binaries["databricks"]: + report["databricks"] = run([binaries["databricks"], "--version"]) + + if args.ug_wheel: + wheel = args.ug_wheel.resolve() + if not wheel.is_file() or wheel.suffix != ".whl": + raise RuntimeError(f"Wheel does not exist: {wheel}") + elif args.ug_version == "checkout": + if not (ROOT / "pyproject.toml").is_file(): + raise RuntimeError( + "No checkout in this image. Pass --ug-version or mount --ug-wheel." + ) + wheels = output / "wheels" + run([uv, "build", "--wheel", "--out-dir", wheels, ROOT], cwd=ROOT) + (wheel,) = wheels.glob("*.whl") + report["git_commit"] = run(["git", "rev-parse", "HEAD"], cwd=ROOT) + report["tracked_diff"] = run( + ["git", "diff", "HEAD", "--", "src", "pyproject.toml"], cwd=ROOT + ) + else: + wheel = None + + if wheel: + report["wheel_sha256"] = hashlib.sha256(wheel.read_bytes()).hexdigest() + package = str(wheel) if wheel else f"ucode=={args.ug_version}" + constraints = output / "requested-constraints.txt" + constraints.write_text( + (args.constraints.read_text() if args.constraints else "") + + "\n" + + "\n".join(args.dependency) + ) + run( + [ + uv, + "pip", + "install", + "--python", + python, + "--index-url", + args.index_url, + "--constraint", + constraints, + package, + ] + ) + run([uv, "pip", "check", "--python", python]) + freeze = run([uv, "pip", "freeze", "--python", python]) + (output / "installed.txt").write_text(freeze + "\n") + (output / "dependencies.txt").write_text( + "\n".join( + line for line in freeze.splitlines() if not re.match(r"ucode(?:==|\s*@)", line) + ) + + "\n" + ) + # The report proves we imported site-packages, not src/ via an editable install. + report["package"] = json.loads( + run( + [ + python, + "-c", + ( + "import importlib.metadata as m, json, ucode; " + "print(json.dumps({'version': m.version('ucode'), 'path': ucode.__file__}))" + ), + ] + ) + ) + package_path = Path(report["package"]["path"]).resolve() + if not package_path.is_relative_to(runtime): + raise RuntimeError( + f"Application was imported outside its isolated environment: {package_path}" + ) + binary = runtime / "bin" / args.entry_point + if not binary.is_file(): + raise RuntimeError( + f"Selected release has no {args.entry_point} entry point; try --entry-point ucode." + ) + + agents = [agent for agent in AGENT_PACKAGES if getattr(args, f"{agent}_version")] + npm_prefix = output / "agents" + npm_prefix.mkdir() + if args.npm_lock: + (npm_prefix / "package.json").write_text( + json.dumps( + { + "dependencies": { + AGENT_PACKAGES[a]: getattr(args, f"{a}_version") for a in agents + } + } + ) + ) + shutil.copyfile(args.npm_lock, npm_prefix / "package-lock.json") + run( + [ + binaries["npm"], + "ci", + "--prefix", + npm_prefix, + "--no-audit", + "--no-fund", + "--registry", + args.npm_registry, + ] + ) + else: + run( + [ + binaries["npm"], + "install", + "--prefix", + npm_prefix, + "--no-audit", + "--no-fund", + "--save-exact", + "--registry", + args.npm_registry, + *[f"{AGENT_PACKAGES[a]}@{getattr(args, f'{a}_version')}" for a in agents], + ] + ) + shutil.copyfile(npm_prefix / "package-lock.json", output / "npm-lock.json") + report["npm_packages"] = json.loads( + run( + [ + binaries["npm"], + "ls", + "--prefix", + npm_prefix, + "--depth=0", + "--json", + ] + ) + )["dependencies"] + agent_bin = npm_prefix / "node_modules/.bin" + # Expose only selected executables to tested programs; other installed + # developer agents cannot be discovered accidentally via inherited PATH. + tool_bin = output / "tools" + tool_bin.mkdir() + for name in ("node", "databricks"): + if binaries[name]: + (tool_bin / name).symlink_to(binaries[name]) + runtime_env = dict(base_env) + runtime_env["PATH"] = os.pathsep.join( + map(str, [runtime / "bin", agent_bin, tool_bin, "/usr/bin", "/bin"]) + ) + report["agents"] = {} + for agent in agents: + version = run([agent_bin / agent, "--version"], env=runtime_env, timeout=30) + expected = getattr(args, f"{agent}_version") + if not re.search(rf"(?=1.0.0. The runner +installs the requested agents into a new npm prefix and ug into a new virtualenv. +Pytest and the PTY/screen libraries (pexpect and pyte) live in a different virtualenv, so they cannot accidentally supply a +missing application dependency. No packages are installed into your existing +agent installations or checkout's `.venv`. +Native live runs refuse existing machine-wide Claude/Codex configuration, which +could override the selected workspace even with a fresh home. Use the container +in that case; the runner never edits or bypasses those managed settings. + +Use the existing e2e workspace and its `DATABRICKS_BEARER` credential. Locally, +`--profile YOUR_PROFILE` can mint a bearer for an explicitly selected profile. +No profile or workspace is selected automatically. + +```bash +export UCODE_TEST_WORKSPACE=https://your-existing-e2e-workspace + +python3 scripts/run_integration.py \ + --ug-version checkout \ + --claude-version 2.1.268 \ + --codex-version 0.154.0 \ + --profile YOUR_PROFILE +``` + +`checkout` builds a wheel and installs it with fresh consumer dependency +resolution. **It does not use `uv.lock`.** This exercises the install path that +caught the tomlkit discrepancy in #496. To reproduce a user's release, pass its +exact distribution version instead, e.g. `--ug-version 0.1.0+f7b4b97`. Use +`--index-url` for the Python index that contains that release and `--npm-registry` +for an npm mirror if public npm is unavailable. Older releases that only +provide the `ucode` command require `--entry-point ucode`. + +Select one agent by providing only its version. Exact agent versions are +required; floating `latest`, caret, and tilde versions are rejected. Normal TUI +boot uses the workspace's configuration and needs no model input. Cases that +exercise explicit model arguments use a real `system.ai` model already discovered +by `ug configure`, recorded in that case's `model.json`. Optional `--claude-model` +and `--codex-model` overrides reproduce a particular model-related failure. + +```bash +# Constrain the suspected dependency while keeping the real CLI and gateway. +python3 scripts/run_integration.py \ + --ug-version checkout --claude-version 2.1.268 --codex-version 0.154.0 \ + --claude-model YOUR_CLAUDE_MODEL --codex-model YOUR_CODEX_MODEL \ + --profile YOUR_PROFILE --dependency tomlkit==0.14.0 \ + -- -k 'app_server or subcommand' +``` + +Repeat with `--dependency tomlkit==0.15.1`, or run without constraints to test +what a new consumer gets today. Several `--dependency` options can be supplied. +Constraints incompatible with the selected ug release fail installation. + +For package-only validation without credentials: + +```bash +python3 scripts/run_integration.py \ + --ug-version checkout --claude-version 2.1.268 --installation-only +``` + +This explicitly selects only the installation checks; it does not claim a live +integration pass. Requested live checks fail when credentials, binaries, models, +or capabilities are missing. There are no capability-based skips or retries of +failed model tasks. A failing historical version should remain a failing result. + +## Coverage and boundaries + +| Area | Automated evidence | +| --- | --- | +| Installed package | Console entry point, version, clean-home status, and missing-configuration error from site-packages | +| Configuration | CLI discovery against the real workspace, repeat setup, preserved user settings, revert and repeat revert | +| Authentication failure | A deliberately invalid bearer is rejected by the real workspace before setup succeeds | +| Codex utility dispatch | Real `app`, `app-server`, `exec`, and `mcp` subcommand help with routing on/off, compared with the selected Codex binary's own output | +| Direct Codex app arguments | An invalid option must reach the real `codex app` parser and preserve its error/exit status, without starting a desktop app | +| Claude utility dispatch | Real `auth` and `mcp` subcommand help with routing on/off | +| Codex app server | A real stdio `initialize` exchange with routing on/off, with and without a launcher `--` separator; JSON must arrive on stdout, separately from stderr diagnostics | +| Agent execution | A real agent reads an unpredictable value from a fixture file and returns it in its structured final response | +| Model options | `--model VALUE`, `--model=VALUE`, and `-m VALUE`, forwarded after `--`, with global routing enabled; none may start a routing wrapper | +| Prompt input | Argument, stdin, and nested `--` forms must complete the same real file-reading task | +| Caller settings | Claude receives a settings path containing spaces, runs the caller's hook, and still authenticates through ug | +| Interactive boot | Real PTY; first agent startup and reopening the same home; routing on/off and explicit-model bypass; visible onboarding, prompt keyboard input, `/exit` and exit status 0 | +| Dependency compatibility | Independent consumer resolution plus explicit constraints; CI exercises tomlkit 0.14.0 and 0.15.1 | + +With both agents selected, there are 41 cases: 3 installation, 4 lifecycle, +18 utility/protocol, 10 headless task cases, and 6 TUI cases. Each TUI case +boots twice (first startup and reopen). The `--` and caller-settings +cases exercise the argument forms used by launchers +such as Isaac. **They do not launch Isaac itself.** Interactive first-prompt +routing, actual desktop app startup, terminal resize/signals, OS-managed settings, +and updater execution are not covered by the current tests. They require +native/PTY scenarios, especially on macOS; `codex app --help` proves dispatch and +config serialization, not that the desktop app opened. New regressions should +add an explicit row/case rather than broaden the meaning of an existing test. + +**TUI fidelity:** `test_tui.py` starts installed `ug claude` / `ug codex` in a real +controlling terminal. After the CLI creates gateway configuration, it walks +through recognized visible agent onboarding/trust screens, requires the TUI's +prompt, types and clears text, exits through `/exit`, and repeats with the state +the agent actually wrote. It also checks that routing wrappers start only in the +expected launch modes and do not route before a model prompt. Unknown onboarding +screens, missing prompts, abnormal exits, and timeouts fail with terminal evidence. +No onboarding state is fabricated. The terminal libraries render ANSI output and +answer terminal-device queries; they do not emulate agent or gateway behavior. + +These boot cases do **not** submit inference requests, cover a second conversation +turn, or exercise tool permission dialogs. Real task execution is separately +tested using Claude `-p` and Codex `exec`. The next TUI cases need a completed +interactive task and successful first-prompt routing. Colima isolates the Linux +environment; native macOS/Windows behavior needs its own runs. + +Run just the TUI cases by adding `-- -m tui` to the runner command. + +Live task tests make inference requests; model overrides can bound their cost. +The remote gateway, its managed settings, and model availability remain external +inputs; this suite is isolated, not an offline emulation of Databricks. + +## Reproduce a failure + +Each run writes a new `.integration-runs//` directory containing: + +- `versions.json`: requested and observed ug/agent versions, Python, Node, uv, + Databricks CLI, platform, source revision/diff, suite hash, and wheel hash when available. +- `dependencies.txt` and `npm-lock.json`: the resolved Python and npm dependency + graphs. Replay them with `--constraints` and `--npm-lock`. +- `test-dependencies.txt`: the separately installed pytest/terminal-tool dependencies. +- `junit.xml`: exact test outcomes and parametrized case names. +- `artifacts/`: command arguments, exit codes, timeout status, redacted output, + app-server protocol diagnostics, and TUI transcripts/rendered screens plus + keystroke actions and routing logs. No credential files are archived. +- `wheels/`: the tested wheel when built from the checkout; replay it with + `--ug-wheel`. For release installations, `installed.txt` records the resolution. + +Per-test homes and working directories are deleted even on failure. +The working directory is outside the checkout so an agent cannot inherit its +project settings or instruction files by walking parent directories. Virtualenvs, +agent packages, and build caches remain under the results directory for local +inspection; remove that run directory when finished. Agent versions are checked +before and after the suite so an automatic upgrade cannot silently change the +combination being tested. Model requests and subprocesses have deadlines, and +the process group is cleaned up after each command. +Selection after `--` accepts `-k`, `-m`, `-x`, and `--maxfail`; configuration and +report paths cannot be overridden. `--installation-only` always restricts the +selection to installation checks, including when additional filters are used. + +## Run in GitHub Actions + +The **Integration** workflow runs on relevant pull requests and pushes to `main`. +Its installation job needs no credentials. For same-repository PRs, the live jobs +reuse the existing `UCODE_TEST_WORKSPACE` and `DATABRICKS_BEARER` secrets. Fork PRs +run installation checks only because they cannot receive those secrets. + +The workspace check requires the secret to match +`https://eng-ml-inference-team-us-east-1.cloud.databricks.com` (a trailing slash +is accepted). It never changes the secret or switches workspaces. There is no CI +model-discovery or model-selection job. Real `ug configure` performs its normal +workspace discovery inside each test; only explicit-model scenarios choose and +record a discovered `system.ai` model as a test argument. +The live matrix covers unconstrained resolution, tomlkit 0.14.0, and tomlkit 0.15.1. +The workflow consumes the stored bearer; it does not mint or refresh credentials. + +For a manual run, use **Actions → Integration → Run workflow**, select the branch, +and choose `all`, `tui`, or `installation`. Set the ug/agent versions. From the CLI: + +```bash +gh workflow run integration.yml -R databricks/unity-gateway --ref YOUR_BRANCH \ + -f suite=tui -f ug_version=checkout \ + -f claude_version=2.1.268 -f codex_version=0.154.0 +gh run list -R databricks/unity-gateway --workflow integration.yml +gh run watch RUN_ID -R databricks/unity-gateway --exit-status +``` + +GitHub enables manual dispatch once the workflow exists on the default branch. +Before this PR merges, its pull-request event runs the workflow. Missing +credentials or a workspace mismatch fail the workspace job. Expired or invalid +credentials fail the actual workspace calls. Those failures do not count as live +test passes. + +## Reproduce and debug a CI failure locally + +Use the same runner and the failing job's artifacts. A new developer machine +needs Python 3.12+, uv, Node/npm, Databricks CLI, and its own authorized login for +the CI workspace. Select that local profile explicitly; CI secrets are not downloaded. + +```bash +gh run download RUN_ID -R databricks/unity-gateway \ + -n integration-live-0 -D .integration-runs/from-ci +``` + +Use `integration-live-1` or `integration-live-2` for the other dependency jobs, +or `integration-installation` for package failures. Read `versions.json` for the +exact agent versions, model overrides, entry point, platform and source revision. +For an explicit-model case without a runner override, read its `model.json` for +the exact model used. Basic boot cases require no model arguments. Use +the archived wheel so a changed checkout cannot alter the reproduction: + +```bash +python3 scripts/run_integration.py \ + --ug-wheel .integration-runs/from-ci/wheels/EXACT_WHEEL.whl \ + --entry-point ug \ + --claude-version CLAUDE_VERSION_FROM_REPORT --codex-version CODEX_VERSION_FROM_REPORT \ + --workspace https://eng-ml-inference-team-us-east-1.cloud.databricks.com \ + --profile YOUR_E2E_PROFILE \ + --constraints .integration-runs/from-ci/dependencies.txt \ + --npm-lock .integration-runs/from-ci/npm-lock.json \ + --output .integration-runs/repro-1 \ + -- -k test_tui_boot_reopen_and_exit +``` + +For an explicit-model failure, also pass the recorded `--claude-model` or +`--codex-model`. For a release run without an archived wheel, use the reported `--ug-version`. +Match Python and Node versions from the report too. `npm-lock.json` replay must +use the same OS/architecture as the original run; add `--platform linux/amd64` +to both `docker build` and `docker run` on an ARM Mac to match GitHub's Ubuntu runner. Changing platforms or +resolving a fresh npm lock is a new comparison, not an exact dependency replay. + +Use `-- -m tui` for all boot cases or `-- -k 'codex and routing-on'` to narrow a +failure. Each rerun needs a new output directory. Inspect: + +- `junit.xml` for the failing case and assertion. +- `artifacts//command-*.json` for the real argv, exit status, stdout and stderr. +- `artifacts//first-boot.json` / `reopen.json` for rendered terminal screens, raw terminal + output, keyboard actions, exit status and routing logs. +- `install.log` for resolution/bootstrap failures. + +Unknown onboarding screens fail with their actual screen text. Update terminal +selectors only after confirming the agent's intended UI changed; do not seed its +onboarding state or relax the prompt/task assertions. Test homes are deleted after +each case; redacted diagnostics remain. For manual interaction, configure a fresh +home with the same installed binaries and recorded public CLI arguments. + +## Colima / Docker + +Colima provides the Linux Docker engine on macOS. The optional image pins the +Python, Node, uv, and Databricks toolchain; the same runner selects ug and agent +versions inside it. Build from the repository root: + +```bash +colima start +COPYFILE_DISABLE=1 tar --format=ustar --exclude=__pycache__ --exclude=.pytest_cache \ + -cf - scripts/run_integration.py tests/integration | \ + docker build -f tests/integration/Dockerfile -t ug-integration - + +# Reuse the same e2e variables. Credentials are passed at runtime, never built +# into the image. The named volume keeps results after the container exits. +docker volume create ug-integration-results +docker run --rm --init \ + -e UCODE_TEST_WORKSPACE -e DATABRICKS_BEARER \ + -v ug-integration-results:/results \ + ug-integration \ + --ug-version YOUR_RELEASE_VERSION \ + --claude-version 2.1.268 --codex-version 0.154.0 \ + -- -m tui +``` + +Use a new results volume for each run, or pass a new `--output /results/NAME`. +To test a checkout, build a wheel on the host (`uv build --wheel`), mount the +wheel directory read-only, and pass `--ug-wheel /wheels/FILE.whl` instead of a +release version. The image deliberately contains no source checkout or host +agent configuration. Record the built image digest when sharing a reproduction; +native runs also depend on the host's OS and toolchain. +The explicit build archive includes only the runner and integration files, even +with legacy Docker builders that ignore per-Dockerfile ignore rules. It also +omits macOS extended attributes that Linux cannot unpack. diff --git a/tests/integration/conftest.py b/tests/integration/conftest.py new file mode 100644 index 000000000..71d0a5064 --- /dev/null +++ b/tests/integration/conftest.py @@ -0,0 +1,90 @@ +"""Standalone fixtures: application state is created only by installed CLI commands.""" + +from __future__ import annotations + +import os +import re +import shutil +import tempfile +from pathlib import Path + +import pytest +from harness import UserSession + + +def pytest_generate_tests(metafunc): + explicit = any( + "agent" + in ( + mark.args[0].replace(" ", "").split(",") + if isinstance(mark.args[0], str) + else mark.args[0] + ) + for mark in metafunc.definition.iter_markers("parametrize") + ) + if "agent" in metafunc.fixturenames and not explicit: + agents = os.environ.get("UG_INTEGRATION_AGENTS", "claude,codex").split(",") + metafunc.parametrize("agent", agents) + + +def pytest_collection_modifyitems(config, items): + agents = os.environ.get("UG_INTEGRATION_AGENTS", "claude,codex").split(",") + selected, deselected = [], [] + for item in items: + # Fail if someone accidentally invokes this under the unit-test fixtures. + if "monkeypatch" in item.fixturenames: + raise pytest.UsageError("Use the integration runner; unit fixtures were inherited.") + if any(item.get_closest_marker(a) and a not in agents for a in ("claude", "codex")): + deselected.append(item) + else: + selected.append(item) + items[:] = selected + config.hook.pytest_deselected(items=deselected) + + +@pytest.fixture(scope="session") +def installed_binary(): + raw = os.environ.get("UG_INTEGRATION_BIN") + if not raw or not Path(raw).is_file(): + pytest.fail("Run scripts/run_integration.py to install the version under test.") + return Path(raw) + + +@pytest.fixture(scope="session") +def workspace(): + value = os.environ.get("UCODE_TEST_WORKSPACE", "").strip().rstrip("/") + if not value.startswith("https://") or not os.environ.get("DATABRICKS_BEARER", "").strip(): + pytest.fail("Live integration requires UCODE_TEST_WORKSPACE and DATABRICKS_BEARER.") + return value + + +@pytest.fixture +def session(request, installed_binary): + # Codex rejects helper installation beneath /tmp. Keep the disposable home + # under the runner's own directory, never in the developer's agent folders. + root = Path(os.environ["UG_INTEGRATION_RUN_DIR"]) + case = re.sub(r"[^a-zA-Z0-9_.-]", "_", request.node.name) + with ( + tempfile.TemporaryDirectory(prefix="case-", dir=root) as temporary, + tempfile.TemporaryDirectory(prefix="ug-integration-project-") as project, + ): + # Agents walk parent directories for project settings. Keeping cwd out + # of the checkout prevents its .claude/AGENTS.md from influencing a run. + yield UserSession( + Path(temporary), Path(project), installed_binary, root / "artifacts" / case + ) + + +@pytest.fixture +def live_session(session, workspace): + for binary in ["databricks", *os.environ["UG_INTEGRATION_AGENTS"].split(",")]: + if not shutil.which(binary, path=session.env["PATH"]): + pytest.fail(f"Required integration binary is missing: {binary}") + session.env["DATABRICKS_BEARER"] = os.environ["DATABRICKS_BEARER"] + return session + + +@pytest.fixture +def configured(live_session, workspace, agent): + live_session.configure(agent, workspace) + return live_session diff --git a/tests/integration/harness.py b/tests/integration/harness.py new file mode 100644 index 000000000..3b53cec9a --- /dev/null +++ b/tests/integration/harness.py @@ -0,0 +1,262 @@ +"""Drive installed programs; never import the application under test.""" + +from __future__ import annotations + +import contextlib +import json +import os +import queue +import re +import signal +import subprocess +import threading +import time +from pathlib import Path + +ANSI = re.compile(r"\x1b\[[0-9;?]*[ -/]*[@-~]") + + +def clean_environment(home: Path) -> dict[str, str]: + # An allowlist prevents a developer's agent keys, settings, plugins, Python + # imports, and routing flags from silently changing the tested combination. + keep = ( + "PATH", + "SYSTEMROOT", + "COMSPEC", + "LANG", + "LC_ALL", + "SSL_CERT_FILE", + "SSL_CERT_DIR", + "REQUESTS_CA_BUNDLE", + "NODE_EXTRA_CA_CERTS", + "HTTPS_PROXY", + "HTTP_PROXY", + "NO_PROXY", + ) + env = {key: os.environ[key] for key in keep if key in os.environ} + env.update( + { + "HOME": str(home), + "USERPROFILE": str(home), + "XDG_CONFIG_HOME": str(home / ".config"), + "XDG_CACHE_HOME": str(home / ".cache"), + "XDG_DATA_HOME": str(home / ".local/share"), + "CLAUDE_CONFIG_DIR": str(home / ".claude"), + "CODEX_HOME": str(home / ".codex"), + "DATABRICKS_CONFIG_FILE": str(home / ".databrickscfg"), + "DISABLE_AUTOUPDATER": "1", + "CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1", + "NO_COLOR": "1", + "TERM": "dumb", + "COLUMNS": "160", + "ENABLE_SMART_ROUTING_V2": "0", + "PYTHONNOUSERSITE": "1", + } + ) + return env + + +def stop_process(proc: subprocess.Popen) -> None: + """Reap the entire process group, including servers left by a failed agent.""" + if os.name == "posix": + with contextlib.suppress(ProcessLookupError): + os.killpg(proc.pid, signal.SIGTERM) + try: + proc.wait(timeout=5) + except subprocess.TimeoutExpired: + pass + # The leader may have exited while a grandchild kept running. + with contextlib.suppress(ProcessLookupError): + os.killpg(proc.pid, signal.SIGKILL) + elif proc.poll() is None: + proc.kill() + proc.wait(timeout=5) + + +class UserSession: + def __init__(self, root: Path, project_root: Path, binary: Path, artifacts: Path): + self.home = root / "home" + self.cwd = project_root / "project with spaces" + self.home.mkdir(parents=True) + self.cwd.mkdir() + self.binary = binary + self.env = clean_environment(self.home) + self.artifacts = artifacts + self.artifacts.mkdir(parents=True, exist_ok=True) + self.commands = 0 + + def redact(self, text: str) -> str: + for token in (os.environ.get("DATABRICKS_BEARER"), self.env.get("DATABRICKS_BEARER")): + if token: + text = text.replace(token, "") + return ANSI.sub("", text) + + def run( + self, + *args: str, + timeout: int = 120, + ok: bool = True, + binary=None, + input_text: str | None = None, + ): + command = [str(binary or self.binary), *args] + proc = subprocess.Popen( + command, + cwd=self.cwd, + env=self.env, + stdin=subprocess.PIPE if input_text is not None else subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + start_new_session=os.name == "posix", + ) + timed_out = False + try: + stdout, stderr = proc.communicate(input=input_text, timeout=timeout) + except subprocess.TimeoutExpired: + timed_out = True + stop_process(proc) + stdout, stderr = proc.communicate(timeout=5) + finally: + stop_process(proc) + result = subprocess.CompletedProcess( + command, proc.returncode, self.redact(stdout), self.redact(stderr) + ) + self.commands += 1 + self.record( + f"command-{self.commands}.json", + { + "argv": command, + "returncode": result.returncode, + "timed_out": timed_out, + "stdin": input_text, + "stdout": result.stdout, + "stderr": result.stderr, + }, + ) + detail = f"{command}\n{result.stdout}\n{result.stderr}" + assert not timed_out, f"Command exceeded {timeout}s:\n{detail}" + if ok: + assert result.returncode == 0, detail + return result + + def record(self, name: str, value: object) -> None: + (self.artifacts / name).write_text(self.redact(json.dumps(value, indent=2))) + + def configure(self, agent: str, workspace: str, *, ok: bool = True): + return self.run( + "configure", + "--agents", + agent, + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ok=ok, + ) + + def state(self) -> dict: + return json.loads((self.home / ".ucode/state.json").read_text()) + + def model_for_explicit_case(self, agent: str) -> str: + """Use a real discovered model only when testing an explicit model option.""" + model = os.environ.get(f"UG_INTEGRATION_{agent.upper()}_MODEL", "").strip() + source = "runner override" + if not model: + state = self.state() + workspace = state["workspaces"][state["current_workspace"]] + models = workspace[f"{agent}_models"] + values = models.values() if isinstance(models, dict) else models + prefix = "system.ai.claude-" if agent == "claude" else "system.ai.gpt-" + model = next((value for value in values if value.startswith(prefix)), "") + source = "ug configure discovery" + assert model, ( + f"ug configure found no system.ai model for {agent}; use --{agent}-model to reproduce a specific model." + ) + self.record("model.json", {"model": model, "source": source}) + return model + + def assert_not_routed(self) -> None: + # These are user-visible diagnostics produced only by routing wrappers. + for name in ("codex-v2-interposer.log", "claude-v2-pty.log"): + assert not (self.home / ".ucode" / name).exists(), f"Unexpected routing: {name}" + + def app_server_handshake(self, args: list[str], timeout: int = 120) -> dict: + """Speak the real Codex stdio protocol and require an initialize response.""" + command = [str(self.binary), "codex", *args] + messages: queue.Queue = queue.Queue() + transcript: list[str] = [] + diagnostics: list[str] = [] + proc = subprocess.Popen( + command, + cwd=self.cwd, + env=self.env, + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + bufsize=1, + start_new_session=os.name == "posix", + ) + + def read_output(): + for line in proc.stdout: + transcript.append(self.redact(line)) + try: + messages.put(json.loads(line)) + except ValueError: + if line.strip(): + messages.put({"protocol_error": self.redact(line)}) + messages.put(None) + + def read_diagnostics(): + for line in proc.stderr: + diagnostics.append(self.redact(line)) + + reader = threading.Thread(target=read_output, daemon=True) + stderr_reader = threading.Thread(target=read_diagnostics, daemon=True) + reader.start() + stderr_reader.start() + try: + proc.stdin.write( + json.dumps( + { + "id": 1, + "method": "initialize", + "params": {"clientInfo": {"name": "ug-integration", "version": "1.0.0"}}, + } + ) + + "\n" + ) + proc.stdin.flush() + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + try: + message = messages.get(timeout=max(0.01, deadline - time.monotonic())) + except queue.Empty: + break + if message is None: + break + assert isinstance(message, dict), message + assert "protocol_error" not in message, ( + "Non-JSON output on the app-server protocol stream: " + str(message) + ) + if isinstance(message, dict) and message.get("id") == 1: + assert "error" not in message, message + assert isinstance(message.get("result"), dict), message + assert message["result"].get("userAgent"), message + proc.stdin.write('{"method":"initialized","params":{}}\n') + proc.stdin.flush() + return message + raise AssertionError("No app-server initialize response:\n" + "".join(transcript)) + finally: + stop_process(proc) + reader.join(timeout=5) + stderr_reader.join(timeout=5) + proc.stdin.close() + proc.stdout.close() + proc.stderr.close() + self.record( + "app-server.json", {"argv": command, "stdout": transcript, "stderr": diagnostics} + ) diff --git a/tests/integration/pytest.ini b/tests/integration/pytest.ini new file mode 100644 index 000000000..ed06ccb80 --- /dev/null +++ b/tests/integration/pytest.ini @@ -0,0 +1,9 @@ +[pytest] +testpaths = . +addopts = --strict-markers --tb=short +markers = + installation: installed-package checks that require no workspace + live: requires the real workspace used by the existing e2e suite + tui: real interactive terminal boot, keyboard input, exit and reopen + claude: only runs when Claude Code is explicitly selected + codex: only runs when Codex is explicitly selected diff --git a/tests/integration/terminal.py b/tests/integration/terminal.py new file mode 100644 index 000000000..5c1b7d878 --- /dev/null +++ b/tests/integration/terminal.py @@ -0,0 +1,192 @@ +"""A real PTY and terminal screen for driving installed interactive agents. + +The screen implements terminal device replies; it never substitutes an agent, +gateway, application function, configuration file, or model response. +""" + +from __future__ import annotations + +import contextlib +import os +import re +import signal +import time +import uuid + +import pexpect +import pyte + + +class TerminalScreen(pyte.Screen): + def __init__(self, columns, lines, send): + super().__init__(columns, lines) + self.send = send + + def write_process_input(self, data): + # Real TUIs query cursor position/device attributes during startup. + # pyte replies according to the terminal state it has actually rendered. + self.send(data) + + +class AgentTerminal: + def __init__(self, session, agent, command, name): + self.session = session + self.agent = agent + self.name = name + self.command = command + env = {**session.env, "TERM": "xterm-256color"} + self.child = pexpect.spawn( + self.command[0], + self.command[1:], + cwd=str(session.cwd), + env=env, + encoding="utf-8", + codec_errors="replace", + dimensions=(40, 140), + timeout=120, + ) + self.screen = TerminalScreen(140, 40, self.child.send) + self.stream = pyte.Stream(self.screen) + self.output = [] + self.actions = [] + self.ended = False + + @property + def visible(self): + return "\n".join(line.rstrip() for line in self.screen.display) + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc, traceback): + # pexpect creates a new session with a controlling terminal. Clean up + # its process group even if the ug leader exited before its children. + with contextlib.suppress(ProcessLookupError): + os.killpg(self.child.pid, signal.SIGTERM) + self.child.close(force=True) + with contextlib.suppress(ProcessLookupError): + os.killpg(self.child.pid, signal.SIGKILL) + routing_log = ( + self.session.home + / ".ucode" + / ("claude-v2-pty.log" if self.agent == "claude" else "codex-v2-interposer.log") + ) + self.session.record( + f"{self.name}.json", + { + "argv": self.command, + "terminal": {"rows": 40, "columns": 140, "term": "xterm-256color"}, + "actions": self.actions, + "transcript": "".join(self.output), + "screen": self.visible, + "exitstatus": self.child.exitstatus, + "signalstatus": self.child.signalstatus, + "normal_exit_observed": self.ended and self.child.exitstatus == 0, + "routing_log": routing_log.read_text() if routing_log.is_file() else None, + }, + ) + + def read(self): + try: + chunk = self.child.read_nonblocking(size=65536, timeout=0.2) + except pexpect.TIMEOUT: + return + except pexpect.EOF: + self.ended = True + return + self.output.append(chunk) + self.stream.feed(chunk) + + def send(self, keys, reason): + self.actions.append({"reason": reason, "keys": keys, "screen_before": self.visible}) + self.child.send(keys) + + def wait_for(self, predicate, description, timeout=30): + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + self.read() + assert not self.ended, f"TUI exited while waiting for {description}:\n{self.visible}" + if predicate(self.visible): + return + raise AssertionError(f"TUI did not show {description} within {timeout}s:\n{self.visible}") + + def boot(self, timeout=120): + """Handle only recognized visible onboarding; unknown screens fail.""" + handled = set() + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + self.read() + assert not self.ended, f"TUI exited before its prompt:\n{self.visible}" + text = self.visible + # These are interactions with ordinary UI choices, not pre-written + # onboarding state. Only the test's disposable project is trusted. + dialogs = [ + ( + "theme", + "Choose the text style" in text and "Dark mode" in text, + "\r", + ), + ( + "security-notes", + "Security notes" in text and "Enter to continue" in text, + "\r", + ), + ( + "trust-folder", + self.session.cwd.name in text + and bool(re.search(r"1[.)]\s+Yes, I trust (?:this|the) folder", text)), + "1\r", + ), + ( + "trust-directory", + self.session.cwd.name in text + and "trust" in text.lower() + and bool(re.search(r"1[.)]\s+Yes, (?:continue|proceed)", text)), + "1\r", + ), + ] + matched = False + for label, shown, keys in dialogs: + if shown: + matched = True + if label not in handled: + self.send(keys, label) + handled.add(label) + break + if matched: + continue + assert "Select login method:" not in text, ( + "Configured ug launched Claude's account-login flow instead of its gateway session:\n" + + text + ) + assert not ("Sign in with ChatGPT" in text and "Provide your own API key" in text), ( + "Configured ug launched Codex's account-login flow instead of its gateway session:\n" + + text + ) + title = "Claude Code" if self.agent == "claude" else "Codex" + if ( + title in text + and re.search(r"\?\s+for\s+shortcuts", text, re.IGNORECASE) + and re.search(r"(?m)^\s*[❯›>]", text) + ): + self.actions.append({"reason": "prompt-ready", "screen": text}) + return + raise AssertionError( + f"TUI did not reach a usable prompt within {timeout}s:\n{self.visible}" + ) + + def check_input_and_exit(self): + marker = "ug-boot-" + uuid.uuid4().hex[:12] + self.send(marker, "type into the prompt without submitting a model request") + self.wait_for(lambda text: marker in text, "typed text in the prompt") + self.send("\x15", "Ctrl-U clears the prompt") + self.wait_for(lambda text: marker not in text, "cleared prompt") + self.send("/exit\r", "exit through the agent's local slash command") + deadline = time.monotonic() + 30 + while not self.ended and time.monotonic() < deadline: + self.read() + assert self.ended, f"TUI did not exit after /exit:\n{self.visible}" + self.child.close(force=False) + assert self.child.exitstatus == 0, ( + f"TUI exit={self.child.exitstatus}, signal={self.child.signalstatus}:\n{self.visible}" + ) diff --git a/tests/integration/test_installation.py b/tests/integration/test_installation.py new file mode 100644 index 000000000..d01deccec --- /dev/null +++ b/tests/integration/test_installation.py @@ -0,0 +1,23 @@ +"""Smoke checks for the installed distribution, outside the source checkout.""" + +import pytest + +pytestmark = pytest.mark.installation + + +def test_console_script(session): + output = session.run("--help").stdout + assert "configure" in output and "revert" in output + assert session.run("--version").stdout.strip() + + +def test_fresh_home_is_unconfigured(session): + assert "Not Configured" in session.run("status").stdout + assert not (session.home / ".ucode/state.json").exists() + + +def test_auth_without_configuration_fails_with_guidance(session): + result = session.run("auth-token", ok=False) + assert result.returncode != 0 + assert "configure" in result.stdout + result.stderr + assert "Traceback" not in result.stdout + result.stderr diff --git a/tests/integration/test_lifecycle.py b/tests/integration/test_lifecycle.py new file mode 100644 index 000000000..b522152a3 --- /dev/null +++ b/tests/integration/test_lifecycle.py @@ -0,0 +1,58 @@ +"""Configure, repeat, and undo real setup; never manufacture ug state.""" + +import json +import tomllib + +import pytest + +pytestmark = pytest.mark.live + + +def test_configure_repeat_and_revert_preserves_user_settings(live_session, workspace, agent): + session = live_session + if agent == "claude": + user_path = session.home / ".claude/settings.json" + user_settings = '{"permissions":{"allow":["Read"]},"env":{"UG_USER_SETTING":"keep"}}\n' + load = json.loads + else: + user_path = session.home / ".codex/config.toml" + user_settings = "# user comment\n[notice]\nhide_rate_limit_model_nudge = true\n" + load = tomllib.loads + user_path.parent.mkdir(parents=True) + user_path.write_text(user_settings) + + session.configure(agent, workspace) + first = session.state() + ws_state = first["workspaces"][workspace] + assert first["current_workspace"] == workspace + assert agent in ws_state["available_tools"] + assert ws_state[f"{agent}_models"], "Discovery returned no models for the selected agent" + assert workspace in session.run("status").stdout + session.configure(agent, workspace) + second = session.state() + assert second["current_workspace"] == workspace + assert second["workspaces"][workspace]["available_tools"] == ws_state["available_tools"] + current = load(user_path.read_text()) + assert all(current.get(key) == value for key, value in load(user_settings).items()) + for path in (session.home / ".ucode").rglob("*"): + if path.is_file(): + assert session.env["DATABRICKS_BEARER"] not in path.read_text(errors="replace"), ( + f"Bearer was persisted in {path.name}" + ) + + session.run("revert") + assert load(user_path.read_text()) == load(user_settings) + assert "Not Configured" in session.run("status").stdout + private = ".claude/ucode-settings.json" if agent == "claude" else ".codex/ucode.config.toml" + assert not (session.home / private).exists() + session.run("revert") # Reverting an already reverted setup is safe. + + +def test_rejected_credentials_do_not_report_success(live_session, workspace, agent): + live_session.env["DATABRICKS_BEARER"] = "ug-integration-intentionally-invalid" + result = live_session.configure(agent, workspace, ok=False) + assert result.returncode != 0 + output = result.stdout + result.stderr + assert "rejected the access token" in output or "401" in output, output + assert "Configuration Complete" not in output + assert not (live_session.home / ".ucode/state.json").exists() diff --git a/tests/integration/test_passthrough.py b/tests/integration/test_passthrough.py new file mode 100644 index 000000000..2c03ad5de --- /dev/null +++ b/tests/integration/test_passthrough.py @@ -0,0 +1,60 @@ +"""Regression cases for #502 and real Codex config serialization (#496).""" + +import pytest + +pytestmark = pytest.mark.live + +# Compare against each real agent's output. A wrapper's own help is not evidence +# that it dispatched the requested subcommand. No updater or desktop app is run. +HELP_CASES = [ + pytest.param("codex", ["app", "--help"], marks=pytest.mark.codex, id="codex-app"), + pytest.param("codex", ["app-server", "--help"], marks=pytest.mark.codex, id="codex-app-server"), + pytest.param("codex", ["exec", "--help"], marks=pytest.mark.codex, id="codex-exec"), + pytest.param("codex", ["mcp", "--help"], marks=pytest.mark.codex, id="codex-mcp"), + pytest.param("claude", ["mcp", "--help"], marks=pytest.mark.claude, id="claude-mcp"), + pytest.param("claude", ["auth", "--help"], marks=pytest.mark.claude, id="claude-auth"), +] + + +@pytest.mark.parametrize("agent,args", HELP_CASES) +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_subcommand_help_reaches_real_agent(live_session, workspace, agent, args, routing): + session = live_session + session.configure(agent, workspace) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + expected = session.run(*args, binary=agent).stdout.strip() + assert expected + actual = session.run(agent, "--", *args).stdout + assert expected in actual, actual + session.assert_not_routed() + + +@pytest.mark.codex +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_codex_app_argument_error_comes_from_real_agent(live_session, workspace, routing): + session = live_session + args = ["app", "--ug-integration-unknown-option"] + expected = session.run(*args, binary="codex", ok=False) + assert expected.returncode != 0 and expected.stderr.strip() + session.configure("codex", workspace) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + # Exercise the direct subcommand form without opening a desktop app or + # allowing ug's own --help option to intercept the request. + actual = session.run("codex", *args, ok=False) + assert actual.returncode == expected.returncode + assert expected.stderr.strip() in actual.stderr, actual.stdout + actual.stderr + session.assert_not_routed() + + +@pytest.mark.codex +@pytest.mark.parametrize("separator", [False, True], ids=["direct", "launcher-separator"]) +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_codex_app_server_protocol(live_session, workspace, separator, routing): + session = live_session + session.configure("codex", workspace) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + args = ["app-server", "--listen", "stdio://"] + if separator: + args.insert(0, "--") + session.app_server_handshake(args) + session.assert_not_routed() diff --git a/tests/integration/test_tasks.py b/tests/integration/test_tasks.py new file mode 100644 index 000000000..5caa98da6 --- /dev/null +++ b/tests/integration/test_tasks.py @@ -0,0 +1,104 @@ +"""Real file-reading tasks, including launcher-style options after `--`.""" + +import json +import uuid + +import pytest + +pytestmark = pytest.mark.live + + +@pytest.mark.parametrize( + "model_form,prompt_form", + [ + pytest.param("separate", "argument", id="model-value"), + pytest.param("equals", "argument", id="model-equals"), + pytest.param("short", "argument", id="model-short"), + pytest.param("separate", "stdin", id="stdin-prompt"), + pytest.param("separate", "separator", id="nested-separator"), + ], +) +def test_agent_reads_file_through_gateway(configured, agent, model_form, prompt_form): + session = configured + model = session.model_for_explicit_case(agent) + nonce = uuid.uuid4().hex + (session.cwd / "input.txt").write_text(nonce + "\n") + prompt = "Read input.txt in the current directory using a tool. Reply with only its contents." + # An explicit model must bypass routing even when globally enabled. + session.env["ENABLE_SMART_ROUTING_V2"] = "1" + model_args = { + "separate": ["--model", model], + "equals": [f"--model={model}"], + "short": ["-m", model], + }[model_form] + input_text = prompt + "\n" if prompt_form == "stdin" else None + # Both separators are real: the outer one belongs to ug, the inner one + # belongs to the selected agent. The prompt remains one argument. + prompt_args = ["--", prompt] if prompt_form == "separator" else [prompt] + if agent == "claude": + # A real caller-supplied settings file exercises the merge needed by + # launchers such as Isaac. Its hook must execute without losing ug auth. + caller_settings = session.cwd / "caller settings.json" + caller_settings.write_text( + json.dumps( + { + "hooks": { + "SessionStart": [ + { + "hooks": [ + { + "type": "command", + "command": "echo caller-hook-ran > caller-hook.txt", + } + ] + } + ] + }, + } + ) + ) + args = [ + "-p", + *model_args, + "--max-turns", + "4", + "--output-format", + "json", + "--allowedTools", + "Read", + "--settings", + str(caller_settings), + *(prompt_args if input_text is None else []), + ] + else: + args = [ + "exec", + "--skip-git-repo-check", + "--json", + *model_args, + *(prompt_args if input_text is None else ["-"]), + ] + result = session.run(agent, "--", *args, timeout=180, input_text=input_text) + # Look in the agent's structured final output, never in the echoed prompt, + # a tool request, or a banner. The nonce was not given to the model. + payloads = [] + for line in result.stdout.splitlines(): + try: + payloads.append(json.loads(line)) + except ValueError: + pass + if agent == "claude": + final = [p for p in payloads if isinstance(p, dict) and p.get("type") == "result"] + assert final and not final[-1].get("is_error"), result.stdout + assert nonce in final[-1].get("result", ""), result.stdout + assert (session.cwd / "caller-hook.txt").read_text().strip() == "caller-hook-ran" + else: + final = [ + p["item"].get("text", "") + for p in payloads + if isinstance(p, dict) + and p.get("type") == "item.completed" + and p.get("item", {}).get("type") == "agent_message" + ] + assert any(nonce in text for text in final), result.stdout + session.assert_not_routed() diff --git a/tests/integration/test_tui.py b/tests/integration/test_tui.py new file mode 100644 index 000000000..fc18c602d --- /dev/null +++ b/tests/integration/test_tui.py @@ -0,0 +1,33 @@ +"""Boot the real interactive terminal, then reopen the state it actually wrote.""" + +import pytest +from terminal import AgentTerminal + +pytestmark = [pytest.mark.live, pytest.mark.tui] + + +@pytest.mark.parametrize("launch", ["routing-off", "routing-on", "explicit-model"]) +def test_tui_boot_reopen_and_exit(configured, agent, launch): + session = configured + session.env["ENABLE_SMART_ROUTING_V2"] = "0" if launch == "routing-off" else "1" + args = [] + if launch == "explicit-model": + model = session.model_for_explicit_case(agent) + args = ["--", "--model", model] + + for name in ("first-boot", "reopen"): + command = [str(session.binary), agent, *args] + with AgentTerminal(session, agent, command, name) as terminal: + terminal.boot() + terminal.check_input_and_exit() + + if launch == "routing-on": + filename = "claude-v2-pty.log" if agent == "claude" else "codex-v2-interposer.log" + path = session.home / ".ucode" / filename + assert path.is_file(), "The real routing wrapper did not start" + log = path.read_text() + session.record("routing-boot.json", {"log": log}) + assert "[READY]" in log, log + assert "[ROUTE]" not in log, "Booting without a model prompt should not route a turn" + else: + session.assert_not_routed() diff --git a/tests/test_integration_contract.py b/tests/test_integration_contract.py new file mode 100644 index 000000000..39c68d411 --- /dev/null +++ b/tests/test_integration_contract.py @@ -0,0 +1,39 @@ +"""Keep the black-box suite independent of application internals and test doubles.""" + +import ast +from pathlib import Path + + +def test_integration_suite_uses_only_public_process_boundaries(): + violations = [] + for path in (Path(__file__).parent / "integration").rglob("*.py"): + for node in ast.walk(ast.parse(path.read_text())): + modules = [] + if isinstance(node, ast.Import): + modules = [alias.name for alias in node.names] + elif isinstance(node, ast.ImportFrom): + modules = [node.module or ""] + if any(module.split(".")[0] in {"ucode", "mock", "unittest"} for module in modules): + violations.append(f"{path.name}:{node.lineno}: imports application or test doubles") + if isinstance(node, ast.Name) and node.id in { + "monkeypatch", + "MonkeyPatch", + "Mock", + "MagicMock", + "patch", + "setattr", + "delattr", + }: + violations.append(f"{path.name}:{node.lineno}: uses {node.id}") + if isinstance(node, ast.Attribute) and node.attr in { + "MonkeyPatch", + "Mock", + "MagicMock", + "mock", + "patch", + "skip", + "skipif", + "xfail", + }: + violations.append(f"{path.name}:{node.lineno}: uses {node.attr}") + assert not violations, "\n".join(violations) From c314f37af6635f7d5b6aa9035d217d922192fbb4 Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Fri, 11 Sep 2026 16:10:10 -0400 Subject: [PATCH 2/6] Organize integration tests around explicit Claude and Codex CUJs --- .github/workflows/integration.yml | 12 +- scripts/run_integration.py | 18 ++- tests/AGENTS.md | 20 +++ tests/README.md | 121 +++++++-------- tests/integration/README.md | 137 +++++++++-------- tests/integration/conftest.py | 36 ++--- tests/integration/evidence.py | 109 ++++++++++++++ tests/integration/harness.py | 12 ++ tests/integration/pytest.ini | 2 + .../test_ug_argument_forwarding.py} | 31 +++- .../test_ug_headless_arguments.py} | 31 +++- .../test_ug_reconfigure.py} | 29 +++- .../test_ug_tui_launch_modes.py} | 19 ++- tests/integration/terminal.py | 141 +++++++++++++++--- tests/integration/test_installation.py | 18 ++- .../integration/test_smart_routing_claude.py | 70 +++++++++ tests/integration/test_smart_routing_codex.py | 70 +++++++++ tests/integration/test_ug_configure_claude.py | 82 ++++++++++ tests/integration/test_ug_configure_codex.py | 76 ++++++++++ tests/test_integration_contract.py | 15 ++ 20 files changed, 860 insertions(+), 189 deletions(-) create mode 100644 tests/integration/evidence.py rename tests/integration/{test_passthrough.py => regressions/test_ug_argument_forwarding.py} (66%) rename tests/integration/{test_tasks.py => regressions/test_ug_headless_arguments.py} (74%) rename tests/integration/{test_lifecycle.py => regressions/test_ug_reconfigure.py} (70%) rename tests/integration/{test_tui.py => regressions/test_ug_tui_launch_modes.py} (64%) create mode 100644 tests/integration/test_smart_routing_claude.py create mode 100644 tests/integration/test_smart_routing_codex.py create mode 100644 tests/integration/test_ug_configure_claude.py create mode 100644 tests/integration/test_ug_configure_codex.py diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index 1ec7e61c5..02ee9a746 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -11,8 +11,8 @@ on: suite: description: Which integration checks to run type: choice - options: [all, tui, installation] - default: all + options: [main, all, tui, regression, installation] + default: main ug_version: description: Exact ucode release, or checkout default: checkout @@ -53,7 +53,7 @@ env: jobs: installation: name: Installation - runs-on: ubuntu-latest + runs-on: ubuntu-22.04 timeout-minutes: 20 steps: - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 @@ -108,9 +108,9 @@ jobs: PY live: - name: Live (${{ matrix.dependency }}) + name: CUJs (${{ matrix.dependency }}) needs: workspace - runs-on: ubuntu-latest + runs-on: ubuntu-22.04 timeout-minutes: 45 strategy: fail-fast: false @@ -120,7 +120,7 @@ jobs: UCODE_TEST_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} DEPENDENCY: ${{ matrix.dependency }} - TEST_MARKER: ${{ inputs.suite == 'tui' && 'tui' || 'live' }} + TEST_MARKER: ${{ inputs.suite == 'all' && 'live' || inputs.suite || 'main' }} steps: - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 with: diff --git a/scripts/run_integration.py b/scripts/run_integration.py index f8cdbcf86..1584e5ffd 100644 --- a/scripts/run_integration.py +++ b/scripts/run_integration.py @@ -68,6 +68,16 @@ def arguments(): parser.add_argument("--codex-version", type=exact_npm_version) parser.add_argument("--claude-model", default=os.environ.get("UG_INTEGRATION_CLAUDE_MODEL")) parser.add_argument("--codex-model", default=os.environ.get("UG_INTEGRATION_CODEX_MODEL")) + parser.add_argument( + "--claude-provider", + default="main.ucode.ci_e2e_anthropic_nonrelay_mps", + help="Existing Anthropic MPS selected in the configure CUJ.", + ) + parser.add_argument( + "--codex-provider", + default="main.ucode.ci_openai_mps", + help="Existing OpenAI MPS selected in the configure CUJ.", + ) parser.add_argument("--python", default=sys.executable, help="Python 3.12+ path or uv version.") parser.add_argument("--dependency", action="append", default=[], metavar="PACKAGE==VERSION") parser.add_argument("--constraints", type=Path, help="Replay a previous dependencies.txt.") @@ -95,7 +105,7 @@ def arguments(): filters.add_argument("--maxfail", type=int) extra = args.pytest_args[1:] if args.pytest_args[:1] == ["--"] else args.pytest_args selected = filters.parse_args(extra) - marker = selected.m + marker = selected.m or ("installation" if args.installation_only else "main") if args.installation_only: marker = f"installation and ({marker})" if marker else "installation" args.pytest_args = [] @@ -226,6 +236,8 @@ def run(command, *, cwd=output, env=base_env, timeout=600) -> str: "codex": args.codex_version, "claude_model": args.claude_model, "codex_model": args.codex_model, + "claude_provider": args.claude_provider, + "codex_provider": args.codex_provider, "dependencies": args.dependency, "workspace": args.workspace, }, @@ -447,6 +459,8 @@ def run(command, *, cwd=output, env=base_env, timeout=600) -> str: "UG_INTEGRATION_BIN": str(binary), "UG_INTEGRATION_RUN_DIR": str(output), "UG_INTEGRATION_AGENTS": ",".join(agents), + "UG_INTEGRATION_CLAUDE_PROVIDER": args.claude_provider, + "UG_INTEGRATION_CODEX_PROVIDER": args.codex_provider, "UCODE_TEST_WORKSPACE": args.workspace or "", "DATABRICKS_BEARER": bearer, } @@ -457,7 +471,7 @@ def run(command, *, cwd=output, env=base_env, timeout=600) -> str: ) suite = ROOT / "tests/integration" suite_hash = hashlib.sha256() - for path in [Path(__file__), *sorted(suite.glob("*.py")), suite / "pytest.ini"]: + for path in [Path(__file__), *sorted(suite.rglob("*.py")), suite / "pytest.ini"]: suite_hash.update(str(path.relative_to(ROOT)).encode() + b"\0" + path.read_bytes()) report["suite_sha256"] = suite_hash.hexdigest() extra = args.pytest_args diff --git a/tests/AGENTS.md b/tests/AGENTS.md index 8a1462bdf..414123e2e 100644 --- a/tests/AGENTS.md +++ b/tests/AGENTS.md @@ -61,6 +61,26 @@ behavior a test claims to exercise. ## Add / modify / remove +### Required integration test format + +- Organize the main suite around complete user journeys, with explicit names + such as `test_ug_configure_claude_databricks` or + `test_smart_routing_codex_first_prompt`. Do not hide the agent/provider behind + generic parametrization in these main tests. +- Every test has a docstring with **Scenario:** and **Expected:**. State what the + user does and the observable evidence required for success, including limits. +- Keep the configure command, launch, task, and assertions visible in the test. + Fixtures provide fresh environments and credentials, never a preconfigured app. +- Helpers may handle processes, terminal keys, transcript parsing, cleanup, and + artifact collection. Do not bury an entire CUJ inside an opaque helper. +- Main configuration journeys must complete a real TUI task. A startup banner, + config file, echoed prompt, or tool output alone does not prove completion. +- Keep existing focused argument/lifecycle checks in `integration/regressions/`. + They remain runnable, with honest failures, separately from `-m main`. +- Current scope is the eight Claude/Codex provider and smart-routing CUJs. + Do not add MCP, skills, tracing, or the broad configure-option matrix without + a new scope request. + - **Add:** state the user scenario and affected versions, choose the category, add a focused test and coverage row. Demonstrate regression failure on the affected combination when it is available. diff --git a/tests/README.md b/tests/README.md index 4cfdcd4ba..f459ed5d0 100644 --- a/tests/README.md +++ b/tests/README.md @@ -1,69 +1,72 @@ -# Test suites and coverage +# Test suites and user journeys -The categories prove different things. A passing unit test or existing e2e test -does not imply that the installed CLI's complete workflow was exercised. +The integration suite runs a freshly installed ug wheel/release, exact real +Claude/Codex versions, and the existing real e2e workspace. It has no application +imports, mocks, monkeypatching, fake binaries/services, or fabricated ug state. -| Category | Location | What is real? | What can be substituted? | Run | -| --- | --- | --- | --- | --- | -| Unit / component | `test_*.py`, excluding the suites below | Function or component under test | Dependencies, subprocesses, network and state paths | `uv run pytest` | -| Existing e2e | `test_e2e.py`, `test_e2e_uc.py`, `test_e2e_tracing.py` | Databricks workspace; some tests launch agents | Several tests patch state/configuration or invoke application helpers directly | `UCODE_TEST_WORKSPACE=… uv run pytest tests/test_e2e.py -v` | -| Existing proxy / wire tests | `test_gateway_proxy_integration.py`, `test_e2e_user_agent.py` | Local HTTP sockets; real agents in User-Agent tests | Local upstream responses, token minting, config/state paths | `uv run pytest tests/test_gateway_proxy_integration.py -v` | -| **Integration** | **`integration/`** | **Installed ug wheel/release, selected real agents, real e2e workspace, real files/processes** | **No application or service behavior: no mocks, monkeypatching, fake servers, or fabricated ug state** | **`python3 scripts/run_integration.py …`** | +| Category | Location | What it proves | +| --- | --- | --- | +| Unit/component | Existing `test_*.py` files | Individual behavior; dependencies may be mocked | +| Existing e2e | `test_e2e*.py` | Real workspace behavior with some patched setup/internal calls | +| Main integration CUJs | `integration/test_ug_configure_*.py`, `integration/test_smart_routing_*.py` | Complete configure → real TUI → task → exit journeys | +| Installation | `integration/test_installation.py` | Fresh installed package without credentials | +| Focused integration regressions | `integration/regressions/` | Public command, argument, protocol, and lifecycle contracts | -Integration has its own pytest configuration and does not inherit the unit -suite's state-patching fixture. Its runner installs ug and pytest into different -environments and does not consume the checkout's `uv.lock`. +## Main CUJ matrix -## New integration coverage matrix +These are **implemented assertions**, not a claim that every agent/version +combination passes. Consult the run's JUnit report and artifacts for results. +Each function states its **Scenario** and **Expected** outcome and shows its +configure and launch commands. No fixture silently configures the application. -**Covered** means an automated assertion exists, not that every version -combination has passed. Consult a run's `junit.xml` and `versions.json` for actual -results. **Not covered** is a gap, not a promise made by a neighboring test. +| Test | Setup and user action | Expected evidence | +| --- | --- | --- | +| `test_ug_configure_claude_databricks` | Configure Claude with Databricks Hosted; open TUI and read a file | Assistant returns an unpredictable file value, exits normally, and reopens with working keyboard input | +| `test_ug_configure_claude_anthropic_mps` | Select the existing Anthropic MPS in the real configure picker; launch without a provider override | Saved provider appears in status; real Claude completes the file task and exits | +| `test_ug_configure_codex_databricks` | Configure Codex with Databricks Hosted; open TUI and read a file | Completed assistant answer contains the file value, normal exit and reopen | +| `test_ug_configure_codex_openai_mps` | Select the existing OpenAI MPS in the real configure picker; launch without a provider override | Saved provider appears in status; real Codex completes the file task and exits | +| `test_smart_routing_claude_first_prompt` | Configure, enable routing, type the first TUI prompt | Real routing decision and prompt replay plus completed task; no routing before submission | +| `test_smart_routing_codex_first_prompt` | Configure, enable routing, type the first TUI prompt | Real routing decision plus completed task; no fallback or routing before submission | +| `test_smart_routing_claude_subagent` | Ask Claude to delegate a file-reading task | Actual child transcript with the answer, correlated routing decision/child start, parent answer | +| `test_smart_routing_codex_subagent` | Ask Codex to delegate a file-reading task | Actual child session with the answer, correlated routing decision/child start, parent answer | -| Behavior / concern | Claude Code | Codex | Test / limitation | -| --- | --- | --- | --- | -| Build/install checkout wheel | Covered | Covered | Runner + `test_installation.py`; imports resolve inside installed runtime | -| User's exact ug and agent versions | Covered | Covered | Release/wheel and exact agent versions; verified before/after execution | -| Consumer dependencies differ from `uv.lock` (#496) | Covered | Covered | Fresh resolution; constraints; CI matrix includes tomlkit 0.14.0 and 0.15.1 | -| Clean-home help, version, status, auth guidance | Covered | Covered | `test_installation.py`; no workspace needed for these checks | -| Configure against real e2e workspace | Covered | Covered | `test_lifecycle.py`; state is produced only by CLI commands | -| Repeat setup; preserve unrelated user settings | Covered | Covered | `test_lifecycle.py` | -| Revert and repeat revert | Covered | Covered | `test_lifecycle.py` | -| Invalid credentials rejected by real service | Covered | Covered | `test_lifecycle.py`; nonzero exit, no successful saved setup | -| Utility dispatch, routing off/on (#502) | `auth`, `mcp` | `app`, `app-server`, `exec`, `mcp` | `test_passthrough.py`; real subcommand help, compared with direct agent invocation | -| Direct `codex app` argument forwarding | Not applicable | Covered | An invalid option must produce the real agent's error and exit status, without opening a desktop app | -| App-server initialization protocol | Not applicable | Covered | Real JSON-RPC on stdout, separate stderr; direct/launcher `--` forms, routing off/on; rejects non-JSON protocol output | -| Agent reads file through real gateway | Covered | Covered | `test_tasks.py`; unpredictable fixture value in structured final answer | -| Explicit model bypasses global routing | Covered | Covered | `--model VALUE`, `--model=VALUE`, and `-m VALUE`; no routing wrapper diagnostics | -| Stdin prompts and nested `--` separators | Covered | Covered | Real file-reading task with prompt piped on stdin or supplied after the agent's own separator | -| Launcher options after `--` | Covered | Covered | Task and app-server cases | -| Caller settings, path with spaces, preserved hook | Covered | Not covered | Claude caller hook executes while gateway authentication still works | -| Interactive TUI boot and reopen after `ug configure` | Covered | Covered | `test_tui.py`; real PTY, first agent startup and persisted-home reopen; routing on/off and explicit-model bypass | -| First ug launch with automatic configuration/upgrades | **Not covered** | **Not covered** | Boot tests explicitly configure ug first; automatic upgrades need a separate version-change scenario | -| TUI prompt keyboard input and normal exit | Covered | Covered | Type and clear an unsubmitted prompt, execute `/exit`, require exit 0; terminal transcripts and rendered screens | -| TUI first-prompt inference / initial prompt after `--` | **Not covered** | **Not covered** | Boot cases submit no model prompt; needs completed interactive tasks and successful first-prompt routing | -| Interactive follow-up prompts and conversation state | **Not covered** | **Not covered** | Headless single-turn tasks do not exercise the TUI's next turn | -| TUI onboarding and project trust | Covered | Covered | Recognized visible dialogs are handled through keystrokes; no seeded onboarding state; unknown screens fail | -| TUI tool permission dialogs | **Not covered** | **Not covered** | Boot cases do not invoke tools or exercise allow/deny decisions | -| Smart routing selects a model | **Not covered** | **Not covered** | Current tests establish when routing must be bypassed | -| Desktop application startup | Not applicable | **Not covered** | `app --help` checks dispatch/serialization, not desktop startup | -| Agent updater execution | **Not covered** | **Not covered** | Needs a separate scenario that intentionally changes versions | -| Isaac starts and completes a session | **Not covered** | **Not covered** | Launcher-style boundaries are tested; Isaac executable is not | -| Native macOS/Windows UI, OS-managed settings, signals/resize | **Not covered** | **Not covered** | Linux container checks do not establish native-platform behavior | -| Workspace switching, token expiry, MCP, skills, tracing | **Not covered here** | **Not covered here** | Existing tests cover some components; no complete new integration journey | -| Gemini, OpenCode, Copilot, Pi, Cursor | **Not covered here** | **Not covered here** | Initial integration scope is Claude and Codex | +The main configuration tests include ug's normal validation. Routing tests use +`--skip-validate` during setup because their own TUI task is the validation. +All keep the requested agent versions with `--skip-upgrade` and disable optional +Databricks AI Tools to keep these basic journeys focused. -See [integration/README.md](integration/README.md) for version selection, -dependency replay, Colima/Docker, CI, and report contents. +## Retained regression coverage -## Maintaining tests +The original focused checks moved into `integration/regressions/`; they were not +deleted when the main suite was narrowed. Select them with `-- -m regression`. -Read [AGENTS.md](AGENTS.md) before adding, changing, or removing tests. Keep this -matrix aligned with actual assertions, including gaps. A regression should name -the broken user command and version combination and exercise the real installed -program. Never replace a broken integration path with an internal function call -or a successful canned response. +| Concern | Coverage | +| --- | --- | +| Fresh consumer resolution differs from `uv.lock` (#496) | Fresh wheel install; dependency constraints/replay; CI tests unconstrained, tomlkit 0.14.0 and 0.15.1 | +| Reconfigure, preserve user settings, revert | CLI-created state and real files; known generated-file cleanup failure remains an assertion | +| Rejected credentials | Real workspace rejection and no successful saved setup | +| Subcommand forwarding (#502) | Real Claude auth/mcp and Codex app/app-server/exec/mcp help, routing off/on | +| Codex app argument error | Real parser error and exit status; excludes unrelated per-launch warnings | +| Codex app-server | Real JSON-RPC initialize, direct/separator forms; non-JSON stdout still fails | +| Headless prompt/model arguments | Real file task, stdin/separators and caller settings/hook; Claude's unsupported `-m` must retain its real error | +| TUI boot modes | Routing off/on/explicit-model, first boot/reopen, input/clear/exit; no inference claim | -The ordinary suite includes `test_integration_contract.py`, which rejects -application imports and common mocking/patching constructs in `integration/`. -It is a guardrail, not a substitute for reviewing what a test actually proves. +## Gaps and deferred scope + +| Scenario | Status / requirement | +| --- | --- | +| MCP and skills CUJs | Deferred at the user's request | +| Broad configure flags, tracing, multiple workspaces, OAuth/PAT flows | Deferred while focusing on basic main CUJs | +| Provider switching, relayed/subscription MPS | Not covered by the four basic provider journeys | +| Initial prompt supplied on the launch command line | Not yet covered by main routing CUJs | +| Follow-up turns and conversation resume | Not covered; reopen proves startup, not conversation resume | +| Exact child model identity | Verified only when the real agent reports it; missing model fields remain unknown in artifacts | +| Full allow/deny tool-permission matrix | Not covered; real onboarding/trust choices are handled through the TUI | +| Desktop Codex app, Isaac itself, auto-upgrades | Not covered by command forwarding or pinned-version tests | +| Native macOS/Windows managed settings, resize/signals | Separate platform coverage needed | +| Other agents | Current main scope is Claude Code and Codex | + +See [integration/README.md](integration/README.md) for commands, CI, artifacts, +and reproduction. Follow [AGENTS.md](AGENTS.md) and [CLAUDE.md](CLAUDE.md) when +adding, modifying, or removing tests. The ordinary suite enforces both the +no-mocking boundary and the Scenario/Expected docstring format. diff --git a/tests/integration/README.md b/tests/integration/README.md index 35f62f67b..c18f25462 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -22,7 +22,7 @@ in that case; the runner never edits or bypasses those managed settings. Use the existing e2e workspace and its `DATABRICKS_BEARER` credential. Locally, `--profile YOUR_PROFILE` can mint a bearer for an explicitly selected profile. -No profile or workspace is selected automatically. +No profile or workspace is selected automatically. The default test selection is `main`. ```bash export UCODE_TEST_WORKSPACE=https://your-existing-e2e-workspace @@ -43,8 +43,7 @@ for an npm mirror if public npm is unavailable. Older releases that only provide the `ucode` command require `--entry-point ucode`. Select one agent by providing only its version. Exact agent versions are -required; floating `latest`, caret, and tilde versions are rejected. Normal TUI -boot uses the workspace's configuration and needs no model input. Cases that +required; floating `latest`, caret, and tilde versions are rejected. Main CUJs use the workspace's configuration and need no model input. Regression cases that exercise explicit model arguments use a real `system.ai` model already discovered by `ug configure`, recorded in that case's `model.json`. Optional `--claude-model` and `--codex-model` overrides reproduce a particular model-related failure. @@ -74,56 +73,61 @@ integration pass. Requested live checks fail when credentials, binaries, models, or capabilities are missing. There are no capability-based skips or retries of failed model tasks. A failing historical version should remain a failing result. -## Coverage and boundaries - -| Area | Automated evidence | -| --- | --- | -| Installed package | Console entry point, version, clean-home status, and missing-configuration error from site-packages | -| Configuration | CLI discovery against the real workspace, repeat setup, preserved user settings, revert and repeat revert | -| Authentication failure | A deliberately invalid bearer is rejected by the real workspace before setup succeeds | -| Codex utility dispatch | Real `app`, `app-server`, `exec`, and `mcp` subcommand help with routing on/off, compared with the selected Codex binary's own output | -| Direct Codex app arguments | An invalid option must reach the real `codex app` parser and preserve its error/exit status, without starting a desktop app | -| Claude utility dispatch | Real `auth` and `mcp` subcommand help with routing on/off | -| Codex app server | A real stdio `initialize` exchange with routing on/off, with and without a launcher `--` separator; JSON must arrive on stdout, separately from stderr diagnostics | -| Agent execution | A real agent reads an unpredictable value from a fixture file and returns it in its structured final response | -| Model options | `--model VALUE`, `--model=VALUE`, and `-m VALUE`, forwarded after `--`, with global routing enabled; none may start a routing wrapper | -| Prompt input | Argument, stdin, and nested `--` forms must complete the same real file-reading task | -| Caller settings | Claude receives a settings path containing spaces, runs the caller's hook, and still authenticates through ug | -| Interactive boot | Real PTY; first agent startup and reopening the same home; routing on/off and explicit-model bypass; visible onboarding, prompt keyboard input, `/exit` and exit status 0 | -| Dependency compatibility | Independent consumer resolution plus explicit constraints; CI exercises tomlkit 0.14.0 and 0.15.1 | - -With both agents selected, there are 41 cases: 3 installation, 4 lifecycle, -18 utility/protocol, 10 headless task cases, and 6 TUI cases. Each TUI case -boots twice (first startup and reopen). The `--` and caller-settings -cases exercise the argument forms used by launchers -such as Isaac. **They do not launch Isaac itself.** Interactive first-prompt -routing, actual desktop app startup, terminal resize/signals, OS-managed settings, -and updater execution are not covered by the current tests. They require -native/PTY scenarios, especially on macOS; `codex app --help` proves dispatch and -config serialization, not that the desktop app opened. New regressions should -add an explicit row/case rather than broaden the meaning of an existing test. - -**TUI fidelity:** `test_tui.py` starts installed `ug claude` / `ug codex` in a real -controlling terminal. After the CLI creates gateway configuration, it walks -through recognized visible agent onboarding/trust screens, requires the TUI's -prompt, types and clears text, exits through `/exit`, and repeats with the state -the agent actually wrote. It also checks that routing wrappers start only in the -expected launch modes and do not route before a model prompt. Unknown onboarding -screens, missing prompts, abnormal exits, and timeouts fail with terminal evidence. -No onboarding state is fabricated. The terminal libraries render ANSI output and -answer terminal-device queries; they do not emulate agent or gateway behavior. - -These boot cases do **not** submit inference requests, cover a second conversation -turn, or exercise tool permission dialogs. Real task execution is separately -tested using Claude `-p` and Codex `exec`. The next TUI cases need a completed -interactive task and successful first-prompt routing. Colima isolates the Linux -environment; native macOS/Windows behavior needs its own runs. - -Run just the TUI cases by adding `-- -m tui` to the runner command. - -Live task tests make inference requests; model overrides can bound their cost. -The remote gateway, its managed settings, and model availability remain external -inputs; this suite is isolated, not an offline emulation of Databricks. +## Main tests and format + +The default selection is the **eight main end-to-end CUJs** in: + +```text +test_ug_configure_claude.py # Databricks Hosted and Anthropic MPS +test_ug_configure_codex.py # Databricks Hosted and OpenAI MPS +test_smart_routing_claude.py # first prompt and real subagent +test_smart_routing_codex.py # first prompt and real subagent +``` + +Each test has a `Scenario:` / `Expected:` docstring and shows its own public +configure command, TUI launch, user task, and assertions. Shared code only handles +process/terminal mechanics, evidence, and cleanup. Fixtures supply an isolated +session and credentials; none manufacture or configure application state. + +Configuration CUJs use normal ug validation, then require their own completed +interactive task. Routing CUJs skip the preliminary validation prompt and require +the actual routed TUI task instead. Tests disable optional Databricks AI Tools and +pass `--skip-upgrade` to preserve the selected version. They use real onboarding +and trust choices, without seeded acceptance or disabled agent sandboxing. + +A fixture file contains an unpredictable value absent from the prompt. Success +requires an assistant answer in the real agent transcript containing that value, +plus normal TUI exit. Codex evidence requires its task-complete event. Subagent +CUJs require a separate child transcript, child answer, and a correlated routing +decision/start event. If an agent omits its child's model, the artifact records +that unknown; the test does not claim exact child model verification. + +MPS CUJs select the existing services already used by e2e: + +- Claude: `main.ucode.ci_e2e_anthropic_nonrelay_mps`. +- Codex: `main.ucode.ci_openai_mps`. + +Use `--claude-provider` / `--codex-provider` to reproduce another existing service. +Those names are recorded in `versions.json`. No service is created or modified. +A missing service or permission fails the selected CUJ, rather than skipping it. + +There are 49 cases with both agents: 8 main CUJs, 3 installation checks, and +38 retained regressions. Choose deliberately: + +```bash +# Append one of these selections to the runner command: +-- -m main # default: eight complete user journeys +-- -m installation # package checks (use --installation-only to need no auth) +-- -m regression # retained argument, lifecycle, and boot checks +-- -m live # main CUJs plus all live regressions +-- -m tui # main CUJs plus focused boot regressions +``` + +The prior generic tests moved under `regressions/` to keep the main journeys easy +to read. Real failures, including generated config left after revert and banners +on app-server stdout, remain assertions in those regressions. The coverage and +gaps matrix is in [../README.md](../README.md). MCP, skills, tracing, the broad +configure-option matrix, and other agents are outside this focused revision. ## Reproduce a failure @@ -141,7 +145,9 @@ Each run writes a new `.integration-runs//` directory containing: - `wheels/`: the tested wheel when built from the checkout; replay it with `--ug-wheel`. For release installations, `installed.txt` records the resolution. -Per-test homes and working directories are deleted even on failure. +Teardown invokes real `ug revert` when setup created state, restoring machine-level +configuration through the public CLI. Per-test homes and working directories are +then deleted even on failure. The working directory is outside the checkout so an agent cannot inherit its project settings or instruction files by walking parent directories. Virtualenvs, agent packages, and build caches remain under the results directory for local @@ -166,15 +172,19 @@ is accepted). It never changes the secret or switches workspaces. There is no CI model-discovery or model-selection job. Real `ug configure` performs its normal workspace discovery inside each test; only explicit-model scenarios choose and record a discovered `system.ai` model as a test argument. -The live matrix covers unconstrained resolution, tomlkit 0.14.0, and tomlkit 0.15.1. +PR CI runs the eight main CUJs in each dependency job; `all` explicitly includes +the retained regressions. The live matrix covers unconstrained resolution, +tomlkit 0.14.0, and tomlkit 0.15.1. Jobs use Ubuntu 22.04; newer Ubuntu runner +policies prevented Codex's bubblewrap tool from reading even the test file in the +first run. The agent sandbox is not disabled or bypassed. The workflow consumes the stored bearer; it does not mint or refresh credentials. For a manual run, use **Actions → Integration → Run workflow**, select the branch, -and choose `all`, `tui`, or `installation`. Set the ug/agent versions. From the CLI: +and choose `main` (default), `all`, `tui`, `regression`, or `installation`. Set the ug/agent versions. From the CLI: ```bash gh workflow run integration.yml -R databricks/unity-gateway --ref YOUR_BRANCH \ - -f suite=tui -f ug_version=checkout \ + -f suite=main -f ug_version=checkout \ -f claude_version=2.1.268 -f codex_version=0.154.0 gh run list -R databricks/unity-gateway --workflow integration.yml gh run watch RUN_ID -R databricks/unity-gateway --exit-status @@ -214,7 +224,7 @@ python3 scripts/run_integration.py \ --constraints .integration-runs/from-ci/dependencies.txt \ --npm-lock .integration-runs/from-ci/npm-lock.json \ --output .integration-runs/repro-1 \ - -- -k test_tui_boot_reopen_and_exit + -- -k test_ug_configure_claude_databricks ``` For an explicit-model failure, also pass the recorded `--claude-model` or @@ -224,13 +234,14 @@ use the same OS/architecture as the original run; add `--platform linux/amd64` to both `docker build` and `docker run` on an ARM Mac to match GitHub's Ubuntu runner. Changing platforms or resolving a fresh npm lock is a new comparison, not an exact dependency replay. -Use `-- -m tui` for all boot cases or `-- -k 'codex and routing-on'` to narrow a -failure. Each rerun needs a new output directory. Inspect: +Use `-- -m main` for the eight CUJs or +`-- -k test_smart_routing_codex_first_prompt` to narrow a failure. Each rerun needs a new output directory. Inspect: - `junit.xml` for the failing case and assertion. - `artifacts//command-*.json` for the real argv, exit status, stdout and stderr. -- `artifacts//first-boot.json` / `reopen.json` for rendered terminal screens, raw terminal - output, keyboard actions, exit status and routing logs. +- `artifacts//first-session.json`, `provider-session.json`, `first-prompt.json`, + `subagent-task.json`, or `reopen.json` for rendered terminal screens, raw terminal + output, keyboard actions, exit status, routing logs, and actual agent-session records. - `install.log` for resolution/bootstrap failures. Unknown onboarding screens fail with their actual screen text. Update terminal @@ -260,7 +271,7 @@ docker run --rm --init \ ug-integration \ --ug-version YOUR_RELEASE_VERSION \ --claude-version 2.1.268 --codex-version 0.154.0 \ - -- -m tui + -- -m main ``` Use a new results volume for each run, or pass a new `--output /results/NAME`. diff --git a/tests/integration/conftest.py b/tests/integration/conftest.py index 71d0a5064..4930430f6 100644 --- a/tests/integration/conftest.py +++ b/tests/integration/conftest.py @@ -12,21 +12,6 @@ from harness import UserSession -def pytest_generate_tests(metafunc): - explicit = any( - "agent" - in ( - mark.args[0].replace(" ", "").split(",") - if isinstance(mark.args[0], str) - else mark.args[0] - ) - for mark in metafunc.definition.iter_markers("parametrize") - ) - if "agent" in metafunc.fixturenames and not explicit: - agents = os.environ.get("UG_INTEGRATION_AGENTS", "claude,codex").split(",") - metafunc.parametrize("agent", agents) - - def pytest_collection_modifyitems(config, items): agents = os.environ.get("UG_INTEGRATION_AGENTS", "claude,codex").split(",") selected, deselected = [], [] @@ -70,9 +55,16 @@ def session(request, installed_binary): ): # Agents walk parent directories for project settings. Keeping cwd out # of the checkout prevents its .claude/AGENTS.md from influencing a run. - yield UserSession( + user = UserSession( Path(temporary), Path(project), installed_binary, root / "artifacts" / case ) + try: + yield user + finally: + # Restore machine-level settings through the same public CLI that + # created them. A later fresh-home test must not inherit this setup. + if (user.home / ".ucode/state.json").is_file(): + user.run("revert") @pytest.fixture @@ -84,7 +76,11 @@ def live_session(session, workspace): return session -@pytest.fixture -def configured(live_session, workspace, agent): - live_session.configure(agent, workspace) - return live_session +@pytest.fixture(scope="session") +def claude_provider(): + return os.environ["UG_INTEGRATION_CLAUDE_PROVIDER"] + + +@pytest.fixture(scope="session") +def codex_provider(): + return os.environ["UG_INTEGRATION_CODEX_PROVIDER"] diff --git a/tests/integration/evidence.py b/tests/integration/evidence.py new file mode 100644 index 000000000..a6684e1b3 --- /dev/null +++ b/tests/integration/evidence.py @@ -0,0 +1,109 @@ +"""Read real agent transcripts and ug routing records without changing them.""" + +import json +import uuid +from pathlib import Path + + +def read_jsonl(path: Path) -> list[dict]: + if not path.is_file(): + return [] + text = path.read_text() + lines = text.splitlines(keepends=True) + records = [] + for index, line in enumerate(lines): + try: + value = json.loads(line) + except json.JSONDecodeError: + # A running agent may not have finished its last write yet. + if index == len(lines) - 1 and not line.endswith("\n"): + break + raise + if isinstance(value, dict): + records.append(value) + return records + + +def agent_sessions(session, agent: str) -> dict[str, list[dict]]: + directory = session.home / (".claude/projects" if agent == "claude" else ".codex/sessions") + return { + str(path.relative_to(directory)): read_jsonl(path) for path in directory.rglob("*.jsonl") + } + + +def assistant_answers(agent: str, records: list[dict]) -> list[str]: + answers = [] + for record in records: + if agent == "claude" and record.get("type") == "assistant": + message = record.get("message", {}) + if message.get("role") == "assistant": + answers.extend( + part["text"] + for part in message.get("content", []) + if part.get("type") == "text" and isinstance(part.get("text"), str) + ) + if agent == "codex" and record.get("type") == "event_msg": + payload = record.get("payload", {}) + if payload.get("type") == "task_complete" and payload.get("last_agent_message"): + answers.append(payload["last_agent_message"]) + return answers + + +def is_child_session(agent: str, path: str, records: list[dict]) -> bool: + if agent == "claude": + return "/subagents/" in path + return any( + record.get("type") == "session_meta" + and isinstance(record.get("payload", {}).get("source"), dict) + and "subagent" in record["payload"]["source"] + for record in records + ) + + +class FileTask: + """Ordinary project input; the expected answer is never included in the prompt.""" + + def __init__(self, session): + self.value = uuid.uuid4().hex + self.filename = "input-" + uuid.uuid4().hex[:8] + ".txt" + (session.cwd / self.filename).write_text(self.value + "\n") + self.prompt = f"Read {self.filename} using a tool. Reply with only its contents." + self.delegate_prompt = ( + f"Delegate this task to one subagent: read {self.filename} using a tool and return " + "its contents. Do not read the file yourself. Wait for the subagent and reply " + "with only the value it returned." + ) + + def completed(self, session, agent: str, *, child: bool = False) -> bool: + for path, records in agent_sessions(session, agent).items(): + if is_child_session(agent, path, records) != child: + continue + if any(self.value in text for text in assistant_answers(agent, records)): + return True + return False + + def assert_completed(self, session, agent: str, *, child: bool = False) -> None: + sessions = agent_sessions(session, agent) + session.record("agent-sessions.json", sessions) + assert self.completed(session, agent, child=child), ( + f"No {'child' if child else 'parent'} assistant answer contained the file's value; " + "echoed prompts and tool results do not count as completed answers." + ) + + +def assert_subagent_routed(session, agent: str) -> None: + """Require a real gateway decision correlated with an actual child start.""" + root = session.home / ".ucode" + decisions = read_jsonl(root / f"{agent}-smart-routing-decisions.jsonl") + audit = read_jsonl(root / f"{agent}-smart-routing-audit.jsonl") + session.record("subagent-routing.json", {"decisions": decisions, "starts": audit}) + assert decisions, "No real subagent routing decision was recorded" + for decision in decisions: + assert decision.get("requested_model") and decision.get("router_model"), decision + decision_ids = {decision["decision_id"] for decision in decisions} + routed_starts = [row for row in audit if row.get("decision_id") in decision_ids] + assert routed_starts and all(row.get("agent_id") for row in routed_starts), audit + assert all(row.get("matches_router_decision") is not False for row in routed_starts), audit + # Some agent versions omit the child's model from SubagentStart. The report + # preserves that unknown value; this test claims decision + spawn + task, + # not model-identity verification when the agent did not expose it. diff --git a/tests/integration/harness.py b/tests/integration/harness.py index 3b53cec9a..c85f60f73 100644 --- a/tests/integration/harness.py +++ b/tests/integration/harness.py @@ -159,6 +159,18 @@ def configure(self, agent: str, workspace: str, *, ok: bool = True): def state(self) -> dict: return json.loads((self.home / ".ucode/state.json").read_text()) + def workspace_state(self) -> dict: + state = self.state() + return state["workspaces"][state["current_workspace"]] + + def routing_log(self, agent: str) -> str: + name = "claude-v2-pty.log" if agent == "claude" else "codex-v2-interposer.log" + path = self.home / ".ucode" / name + assert path.is_file(), f"No routing log was written: {name}" + value = path.read_text() + self.record(name + ".json", {"log": value}) + return value + def model_for_explicit_case(self, agent: str) -> str: """Use a real discovered model only when testing an explicit model option.""" model = os.environ.get(f"UG_INTEGRATION_{agent.upper()}_MODEL", "").strip() diff --git a/tests/integration/pytest.ini b/tests/integration/pytest.ini index ed06ccb80..577c7538d 100644 --- a/tests/integration/pytest.ini +++ b/tests/integration/pytest.ini @@ -2,6 +2,8 @@ testpaths = . addopts = --strict-markers --tb=short markers = + main: complete configuration and smart-routing user journeys + regression: focused command and lifecycle regression checks installation: installed-package checks that require no workspace live: requires the real workspace used by the existing e2e suite tui: real interactive terminal boot, keyboard input, exit and reopen diff --git a/tests/integration/test_passthrough.py b/tests/integration/regressions/test_ug_argument_forwarding.py similarity index 66% rename from tests/integration/test_passthrough.py rename to tests/integration/regressions/test_ug_argument_forwarding.py index 2c03ad5de..d98c91e67 100644 --- a/tests/integration/test_passthrough.py +++ b/tests/integration/regressions/test_ug_argument_forwarding.py @@ -2,7 +2,7 @@ import pytest -pytestmark = pytest.mark.live +pytestmark = [pytest.mark.live, pytest.mark.regression] # Compare against each real agent's output. A wrapper's own help is not evidence # that it dispatched the requested subcommand. No updater or desktop app is run. @@ -18,7 +18,11 @@ @pytest.mark.parametrize("agent,args", HELP_CASES) @pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_subcommand_help_reaches_real_agent(live_session, workspace, agent, args, routing): +def test_ug_subcommand_help_reaches_real_agent(live_session, workspace, agent, args, routing): + """Scenario: request agent subcommand help through ug with routing off/on. + + Expected: the real agent's help is returned and no routing wrapper starts. + """ session = live_session session.configure(agent, workspace) session.env["ENABLE_SMART_ROUTING_V2"] = routing @@ -31,25 +35,38 @@ def test_subcommand_help_reaches_real_agent(live_session, workspace, agent, args @pytest.mark.codex @pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_codex_app_argument_error_comes_from_real_agent(live_session, workspace, routing): +def test_ug_codex_app_preserves_unknown_argument_error(live_session, workspace, routing): + """Scenario: pass an unknown option to ug codex app. + + Expected: the real Codex parser's error and exit code survive forwarding. + Per-launch helper warnings are not part of the argument-error contract. + """ session = live_session args = ["app", "--ug-integration-unknown-option"] - expected = session.run(*args, binary="codex", ok=False) - assert expected.returncode != 0 and expected.stderr.strip() session.configure("codex", workspace) + expected = session.run(*args, binary="codex", ok=False) + assert expected.returncode != 0 and "error:" in expected.stderr session.env["ENABLE_SMART_ROUTING_V2"] = routing # Exercise the direct subcommand form without opening a desktop app or # allowing ug's own --help option to intercept the request. actual = session.run("codex", *args, ok=False) assert actual.returncode == expected.returncode - assert expected.stderr.strip() in actual.stderr, actual.stdout + actual.stderr + parser_error = expected.stderr[expected.stderr.index("error:") :].strip() + assert parser_error in actual.stderr, actual.stdout + actual.stderr session.assert_not_routed() @pytest.mark.codex @pytest.mark.parametrize("separator", [False, True], ids=["direct", "launcher-separator"]) @pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_codex_app_server_protocol(live_session, workspace, separator, routing): +def test_ug_codex_app_server_initializes_over_clean_json_rpc( + live_session, workspace, separator, routing +): + """Scenario: connect a real client to ug codex app-server over stdio. + + Expected: initialize returns a valid response on stdout with no non-JSON + banners mixed into the protocol; direct and launcher separator forms work. + """ session = live_session session.configure("codex", workspace) session.env["ENABLE_SMART_ROUTING_V2"] = routing diff --git a/tests/integration/test_tasks.py b/tests/integration/regressions/test_ug_headless_arguments.py similarity index 74% rename from tests/integration/test_tasks.py rename to tests/integration/regressions/test_ug_headless_arguments.py index 5caa98da6..7d8f8c977 100644 --- a/tests/integration/test_tasks.py +++ b/tests/integration/regressions/test_ug_headless_arguments.py @@ -5,9 +5,16 @@ import pytest -pytestmark = pytest.mark.live +pytestmark = [pytest.mark.live, pytest.mark.regression] +@pytest.mark.parametrize( + "agent", + [ + pytest.param("claude", marks=pytest.mark.claude), + pytest.param("codex", marks=pytest.mark.codex), + ], +) @pytest.mark.parametrize( "model_form,prompt_form", [ @@ -18,8 +25,16 @@ pytest.param("separate", "separator", id="nested-separator"), ], ) -def test_agent_reads_file_through_gateway(configured, agent, model_form, prompt_form): - session = configured +def test_ug_headless_task_accepts_model_and_prompt_arguments( + live_session, workspace, agent, model_form, prompt_form +): + """Scenario: forward explicit model flags and argument/stdin prompts through ug. + + Expected: a real headless agent returns an unknown file value; Claude also + runs the caller's hook. Unsupported options preserve the real agent error. + """ + session = live_session + session.configure(agent, workspace) model = session.model_for_explicit_case(agent) nonce = uuid.uuid4().hex (session.cwd / "input.txt").write_text(nonce + "\n") @@ -78,6 +93,16 @@ def test_agent_reads_file_through_gateway(configured, agent, model_form, prompt_ *model_args, *(prompt_args if input_text is None else ["-"]), ] + if agent == "claude" and model_form == "short": + # Claude 2.1.268 has no -m option. Verify forwarding against the actual + # binary instead of inventing support that the upstream CLI lacks. + expected = session.run(*args, binary=agent, ok=False, input_text=input_text) + actual = session.run(agent, "--", *args, ok=False, input_text=input_text) + assert expected.returncode != 0 and "unknown option '-m'" in expected.stderr + assert actual.returncode == expected.returncode + assert "unknown option '-m'" in actual.stderr + session.assert_not_routed() + return result = session.run(agent, "--", *args, timeout=180, input_text=input_text) # Look in the agent's structured final output, never in the echoed prompt, # a tool request, or a banner. The nonce was not given to the model. diff --git a/tests/integration/test_lifecycle.py b/tests/integration/regressions/test_ug_reconfigure.py similarity index 70% rename from tests/integration/test_lifecycle.py rename to tests/integration/regressions/test_ug_reconfigure.py index b522152a3..08db3214b 100644 --- a/tests/integration/test_lifecycle.py +++ b/tests/integration/regressions/test_ug_reconfigure.py @@ -5,10 +5,22 @@ import pytest -pytestmark = pytest.mark.live +pytestmark = [pytest.mark.live, pytest.mark.regression] -def test_configure_repeat_and_revert_preserves_user_settings(live_session, workspace, agent): +@pytest.mark.parametrize( + "agent", + [ + pytest.param("claude", marks=pytest.mark.claude), + pytest.param("codex", marks=pytest.mark.codex), + ], +) +def test_ug_reconfigure_and_revert_preserve_user_settings(live_session, workspace, agent): + """Scenario: configure twice over existing user settings, then revert twice. + + Expected: unrelated settings survive and ug's generated config is removed. + This is a lifecycle check; the main CUJs separately prove task completion. + """ session = live_session if agent == "claude": user_path = session.home / ".claude/settings.json" @@ -48,7 +60,18 @@ def test_configure_repeat_and_revert_preserves_user_settings(live_session, works session.run("revert") # Reverting an already reverted setup is safe. -def test_rejected_credentials_do_not_report_success(live_session, workspace, agent): +@pytest.mark.parametrize( + "agent", + [ + pytest.param("claude", marks=pytest.mark.claude), + pytest.param("codex", marks=pytest.mark.codex), + ], +) +def test_ug_configure_rejects_invalid_workspace_credentials(live_session, workspace, agent): + """Scenario: configure against the real workspace with an invalid bearer. + + Expected: the service rejects authentication and ug saves no successful setup. + """ live_session.env["DATABRICKS_BEARER"] = "ug-integration-intentionally-invalid" result = live_session.configure(agent, workspace, ok=False) assert result.returncode != 0 diff --git a/tests/integration/test_tui.py b/tests/integration/regressions/test_ug_tui_launch_modes.py similarity index 64% rename from tests/integration/test_tui.py rename to tests/integration/regressions/test_ug_tui_launch_modes.py index fc18c602d..d5aaf9f7d 100644 --- a/tests/integration/test_tui.py +++ b/tests/integration/regressions/test_ug_tui_launch_modes.py @@ -3,12 +3,25 @@ import pytest from terminal import AgentTerminal -pytestmark = [pytest.mark.live, pytest.mark.tui] +pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.regression] @pytest.mark.parametrize("launch", ["routing-off", "routing-on", "explicit-model"]) -def test_tui_boot_reopen_and_exit(configured, agent, launch): - session = configured +@pytest.mark.parametrize( + "agent", + [ + pytest.param("claude", marks=pytest.mark.claude), + pytest.param("codex", marks=pytest.mark.codex), + ], +) +def test_ug_tui_launch_mode_supports_boot_reopen_and_exit(live_session, workspace, agent, launch): + """Scenario: boot and reopen a configured agent in each routing launch mode. + + Expected: onboarding, keyboard editing, and normal exit work; wrappers start + only when requested. This regression makes no inference-completion claim. + """ + session = live_session + session.configure(agent, workspace) session.env["ENABLE_SMART_ROUTING_V2"] = "0" if launch == "routing-off" else "1" args = [] if launch == "explicit-model": diff --git a/tests/integration/terminal.py b/tests/integration/terminal.py index 5c1b7d878..f29bf178a 100644 --- a/tests/integration/terminal.py +++ b/tests/integration/terminal.py @@ -15,6 +15,7 @@ import pexpect import pyte +from evidence import agent_sessions class TerminalScreen(pyte.Screen): @@ -28,7 +29,7 @@ def write_process_input(self, data): self.send(data) -class AgentTerminal: +class TerminalProcess: def __init__(self, session, agent, command, name): self.session = session self.agent = agent @@ -42,10 +43,10 @@ def __init__(self, session, agent, command, name): env=env, encoding="utf-8", codec_errors="replace", - dimensions=(40, 140), + dimensions=(60, 140), timeout=120, ) - self.screen = TerminalScreen(140, 40, self.child.send) + self.screen = TerminalScreen(140, 60, self.child.send) self.stream = pyte.Stream(self.screen) self.output = [] self.actions = [] @@ -75,7 +76,7 @@ def __exit__(self, exc_type, exc, traceback): f"{self.name}.json", { "argv": self.command, - "terminal": {"rows": 40, "columns": 140, "term": "xterm-256color"}, + "terminal": {"rows": 60, "columns": 140, "term": "xterm-256color"}, "actions": self.actions, "transcript": "".join(self.output), "screen": self.visible, @@ -83,6 +84,9 @@ def __exit__(self, exc_type, exc, traceback): "signalstatus": self.child.signalstatus, "normal_exit_observed": self.ended and self.child.exitstatus == 0, "routing_log": routing_log.read_text() if routing_log.is_file() else None, + "agent_sessions": agent_sessions(self.session, self.agent) + if self.agent in ("claude", "codex") + else {}, }, ) @@ -101,23 +105,112 @@ def send(self, keys, reason): self.actions.append({"reason": reason, "keys": keys, "screen_before": self.visible}) self.child.send(keys) - def wait_for(self, predicate, description, timeout=30): + def wait_for(self, predicate, description, timeout=30, stable_for=0.3): deadline = time.monotonic() + timeout + since = None while time.monotonic() < deadline: self.read() assert not self.ended, f"TUI exited while waiting for {description}:\n{self.visible}" if predicate(self.visible): - return + since = since or time.monotonic() + if time.monotonic() - since >= stable_for: + return + else: + since = None raise AssertionError(f"TUI did not show {description} within {timeout}s:\n{self.visible}") + def selected_line(self): + return next( + (line.strip() for line in self.visible.splitlines() if re.match(r"^\s*[›❯>]", line)), + "", + ) + + def choose(self, prompt, label): + """Navigate the visible menu with arrow keys; never write its saved state.""" + self.wait_for(lambda text: prompt in text and self.selected_line(), prompt, timeout=120) + visited = set() + for _ in range(100): + current = self.selected_line() + if label in current: + self.send("\r", f"choose {label}") + self.wait_for( + lambda text, before=current: ( + prompt not in text or self.selected_line() != before + ), + f"confirmation of {label}", + ) + return + assert current not in visited, f"Menu does not offer {label}:\n{self.visible}" + visited.add(current) + self.send("\x1b[B", f"move towards {label}") + self.wait_for( + lambda text, before=current: self.selected_line() != before, "next menu option" + ) + raise AssertionError(f"Could not select {label}") + + def submit(self, text): + self.send(text, "type text before pressing Enter") + compact = "".join(text.split()) + self.wait_for( + lambda screen: compact in "".join(screen.split()), "typed input", stable_for=0.5 + ) + self.send("\r", "press Enter after the input has rendered") + + def finish(self, timeout=120): + deadline = time.monotonic() + timeout + while not self.ended and time.monotonic() < deadline: + self.read() + assert self.ended, f"Process did not exit within {timeout}s:\n{self.visible}" + self.child.close(force=False) + assert self.child.exitstatus == 0, ( + f"exit={self.child.exitstatus}, signal={self.child.signalstatus}:\n{self.visible}" + ) + + +class ConfigureTerminal(TerminalProcess): + def select_agent(self, display): + prompt = "Select coding agents to configure:" + self.wait_for(lambda text: prompt in text and self.selected_line(), prompt, timeout=120) + visited = set() + selected = False + while True: + current = self.selected_line() + match = re.search(r"[›❯>]\s*([●○])\s*(.+)", current) + assert match, f"Unrecognized agent checkbox: {current}" + checked, name = match.groups() + if name in visited: + break + visited.add(name) + wanted = name.strip() == display + selected = selected or wanted + if (checked == "●") != wanted: + self.send(" ", f"{'select' if wanted else 'deselect'} {name}") + self.wait_for( + lambda text, before=current: self.selected_line() != before, "checkbox change" + ) + current = self.selected_line() + self.send("\x1b[B", "next agent checkbox") + self.wait_for(lambda text, before=current: self.selected_line() != before, "next agent") + assert selected, f"{display} was not available in the real agent picker" + self.send("\r", f"configure only {display}") + + +class AgentTerminal(TerminalProcess): def boot(self, timeout=120): """Handle only recognized visible onboarding; unknown screens fail.""" handled = set() + ready_since = None deadline = time.monotonic() + timeout while time.monotonic() < deadline: self.read() assert not self.ended, f"TUI exited before its prompt:\n{self.visible}" text = self.visible + if "Accessing workspace:" in text and self.session.cwd.name in text: + self.choose("Accessing workspace:", "Yes, I trust this folder") + continue + if "Hooks need review" in text: + self.choose("Hooks need review", "Trust all and continue") + continue # These are interactions with ordinary UI choices, not pre-written # onboarding state. Only the test's disposable project is trusted. dialogs = [ @@ -135,14 +228,14 @@ def boot(self, timeout=120): "trust-folder", self.session.cwd.name in text and bool(re.search(r"1[.)]\s+Yes, I trust (?:this|the) folder", text)), - "1\r", + "\r", ), ( "trust-directory", self.session.cwd.name in text and "trust" in text.lower() and bool(re.search(r"1[.)]\s+Yes, (?:continue|proceed)", text)), - "1\r", + "\r", ), ] matched = False @@ -166,11 +259,15 @@ def boot(self, timeout=120): title = "Claude Code" if self.agent == "claude" else "Codex" if ( title in text - and re.search(r"\?\s+for\s+shortcuts", text, re.IGNORECASE) - and re.search(r"(?m)^\s*[❯›>]", text) + and "loading" not in text.lower() + and re.search(r"(?m)^\s*[❯›>]\s*(?!\d+[.)])", text) ): - self.actions.append({"reason": "prompt-ready", "screen": text}) - return + ready_since = ready_since or time.monotonic() + if time.monotonic() - ready_since >= 1: + self.actions.append({"reason": "prompt-ready", "screen": text}) + return + else: + ready_since = None raise AssertionError( f"TUI did not reach a usable prompt within {timeout}s:\n{self.visible}" ) @@ -181,12 +278,16 @@ def check_input_and_exit(self): self.wait_for(lambda text: marker in text, "typed text in the prompt") self.send("\x15", "Ctrl-U clears the prompt") self.wait_for(lambda text: marker not in text, "cleared prompt") - self.send("/exit\r", "exit through the agent's local slash command") - deadline = time.monotonic() + 30 - while not self.ended and time.monotonic() < deadline: - self.read() - assert self.ended, f"TUI did not exit after /exit:\n{self.visible}" - self.child.close(force=False) - assert self.child.exitstatus == 0, ( - f"TUI exit={self.child.exitstatus}, signal={self.child.signalstatus}:\n{self.visible}" + self.exit_normally() + + def wait_for_task(self, task, timeout=180): + self.wait_for( + lambda text: task.completed(self.session, self.agent), + "a completed assistant answer with the file's value", + timeout=timeout, ) + task.assert_completed(self.session, self.agent) + + def exit_normally(self): + self.submit("/exit") + self.finish(timeout=30) diff --git a/tests/integration/test_installation.py b/tests/integration/test_installation.py index d01deccec..2f8312d4b 100644 --- a/tests/integration/test_installation.py +++ b/tests/integration/test_installation.py @@ -5,18 +5,30 @@ pytestmark = pytest.mark.installation -def test_console_script(session): +def test_ug_installed_wheel_exposes_help_and_version(session): + """Scenario: invoke the freshly installed ug console script. + + Expected: public commands are listed and the installed version is reported. + """ output = session.run("--help").stdout assert "configure" in output and "revert" in output assert session.run("--version").stdout.strip() -def test_fresh_home_is_unconfigured(session): +def test_ug_status_in_fresh_home_is_unconfigured(session): + """Scenario: inspect ug before any setup in a fresh home. + + Expected: status reports Not Configured and does not create saved state. + """ assert "Not Configured" in session.run("status").stdout assert not (session.home / ".ucode/state.json").exists() -def test_auth_without_configuration_fails_with_guidance(session): +def test_ug_auth_without_configuration_explains_how_to_configure(session): + """Scenario: request an auth token before configuring ug. + + Expected: nonzero exit and configure guidance, without a Python traceback. + """ result = session.run("auth-token", ok=False) assert result.returncode != 0 assert "configure" in result.stdout + result.stderr diff --git a/tests/integration/test_smart_routing_claude.py b/tests/integration/test_smart_routing_claude.py new file mode 100644 index 000000000..ac6ced5e7 --- /dev/null +++ b/tests/integration/test_smart_routing_claude.py @@ -0,0 +1,70 @@ +"""Main CUJs: real first-prompt and child-task routing in Claude's TUI.""" + +import pytest +from evidence import FileTask, assert_subagent_routed +from terminal import AgentTerminal + +pytestmark = [pytest.mark.live, pytest.mark.main, pytest.mark.tui, pytest.mark.claude] + + +def test_smart_routing_claude_first_prompt(live_session, workspace): + """Scenario: enable smart routing and submit Claude's first interactive prompt. + + Expected: a real gateway decision selects a model, the prompt is replayed, + and the assistant completes the task before a normal exit. Boot alone fails. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + command = [str(session.binary), "claude", "--enable-smart-routing"] + with AgentTerminal(session, "claude", command, "first-prompt") as tui: + tui.boot() + assert "[ROUTE]" not in session.routing_log("claude"), "Boot must not route a prompt" + tui.submit(task.prompt) + tui.wait_for_task(task) + log = session.routing_log("claude") + assert "[ROUTE] first prompt ->" in log, log + assert "[REPLAY] first prompt submitted" in log, log + tui.exit_normally() + task.assert_completed(session, "claude") + + +def test_smart_routing_claude_subagent(live_session, workspace): + """Scenario: ask Claude to delegate a file-reading task with routing enabled. + + Expected: a real child session has a correlated gateway routing decision, + the child returns the file value, and the parent returns that result. + Child model identity remains explicitly unknown if its event omits it. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + command = [str(session.binary), "claude", "--enable-smart-routing"] + with AgentTerminal(session, "claude", command, "subagent-task") as tui: + tui.boot() + tui.submit(task.delegate_prompt) + tui.wait_for_task(task, timeout=240) + task.assert_completed(session, "claude", child=True) + assert_subagent_routed(session, "claude") + tui.exit_normally() + task.assert_completed(session, "claude") diff --git a/tests/integration/test_smart_routing_codex.py b/tests/integration/test_smart_routing_codex.py new file mode 100644 index 000000000..c3bfd712c --- /dev/null +++ b/tests/integration/test_smart_routing_codex.py @@ -0,0 +1,70 @@ +"""Main CUJs: real first-prompt and child-task routing in Codex's TUI.""" + +import pytest +from evidence import FileTask, assert_subagent_routed +from terminal import AgentTerminal + +pytestmark = [pytest.mark.live, pytest.mark.main, pytest.mark.tui, pytest.mark.codex] + + +def test_smart_routing_codex_first_prompt(live_session, workspace): + """Scenario: enable smart routing and submit Codex's first interactive prompt. + + Expected: a real gateway decision selects a model and Codex completes the + file-reading task before a normal exit. A routing fallback does not pass. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + command = [str(session.binary), "codex", "--enable-smart-routing"] + with AgentTerminal(session, "codex", command, "first-prompt") as tui: + tui.boot() + assert "[ROUTE]" not in session.routing_log("codex"), "Boot must not route a prompt" + tui.submit(task.prompt) + tui.wait_for_task(task) + log = session.routing_log("codex") + assert "[ROUTE] selected" in log, log + assert "selection failed" not in log and "settings/update timed out" not in log, log + tui.exit_normally() + task.assert_completed(session, "codex") + + +def test_smart_routing_codex_subagent(live_session, workspace): + """Scenario: ask Codex to delegate a file-reading task with routing enabled. + + Expected: a real child session has a correlated gateway routing decision, + the child returns the file value, and the parent returns that result. + Child model identity remains explicitly unknown if its event omits it. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + command = [str(session.binary), "codex", "--enable-smart-routing"] + with AgentTerminal(session, "codex", command, "subagent-task") as tui: + tui.boot() + tui.submit(task.delegate_prompt) + tui.wait_for_task(task, timeout=240) + task.assert_completed(session, "codex", child=True) + assert_subagent_routed(session, "codex") + tui.exit_normally() + task.assert_completed(session, "codex") diff --git a/tests/integration/test_ug_configure_claude.py b/tests/integration/test_ug_configure_claude.py new file mode 100644 index 000000000..e57488463 --- /dev/null +++ b/tests/integration/test_ug_configure_claude.py @@ -0,0 +1,82 @@ +"""Main CUJs: configure Claude through ug, then use its real interactive session.""" + +import pytest +from evidence import FileTask +from terminal import AgentTerminal, ConfigureTerminal + +pytestmark = [pytest.mark.live, pytest.mark.main, pytest.mark.tui, pytest.mark.claude] + + +def test_ug_configure_claude_databricks(live_session, workspace): + """Scenario: configure Claude with Databricks Hosted and use its TUI. + + Expected: configure succeeds, Claude returns a file value through the real + gateway, exits normally, and can reopen the configuration ug created. + Optional AI Tools are disabled; the selected agent version is kept pinned. + """ + session = live_session + task = FileTask(session) + + # Configure using the installed public CLI, including its normal validation. + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-upgrade", + "--disable-databricks-ai-tools", + timeout=240, + ) + assert not session.workspace_state().get("provider_services", {}).get("claude") + + # Use the real TUI; a config file or startup banner alone is not success. + with AgentTerminal(session, "claude", [str(session.binary), "claude"], "first-session") as tui: + tui.boot() + tui.submit(task.prompt) + tui.wait_for_task(task) + tui.exit_normally() + task.assert_completed(session, "claude") + session.assert_not_routed() + + # Reopen the same home, without configuring again or seeding onboarding state. + with AgentTerminal(session, "claude", [str(session.binary), "claude"], "reopen") as tui: + tui.boot() + tui.check_input_and_exit() + + +def test_ug_configure_claude_anthropic_mps(live_session, workspace, claude_provider): + """Scenario: choose the real Anthropic MPS in ug configure's provider picker. + + Expected: ug saves that provider, and launching Claude without --provider + uses the saved choice to complete a file-reading task and exit normally. + """ + session = live_session + task = FileTask(session) + + # Provider selection is interactive; --agents would bypass this real picker. + command = [ + str(session.binary), + "configure", + "--workspaces", + workspace, + "--skip-upgrade", + "--disable-databricks-ai-tools", + ] + with ConfigureTerminal(session, "claude", command, "configure-provider") as configure: + configure.select_agent("Claude Code") + configure.choose("How should Claude Code get its models?", "External Models") + configure.choose("Select a model provider service:", claude_provider) + configure.finish(timeout=240) + assert session.workspace_state()["provider_services"]["claude"] == claude_provider + assert claude_provider in session.run("status").stdout + + with AgentTerminal( + session, "claude", [str(session.binary), "claude"], "provider-session" + ) as tui: + tui.boot() + tui.submit(task.prompt) + tui.wait_for_task(task) + tui.exit_normally() + task.assert_completed(session, "claude") + session.assert_not_routed() diff --git a/tests/integration/test_ug_configure_codex.py b/tests/integration/test_ug_configure_codex.py new file mode 100644 index 000000000..fe0513d0c --- /dev/null +++ b/tests/integration/test_ug_configure_codex.py @@ -0,0 +1,76 @@ +"""Main CUJs: configure Codex through ug, then use its real interactive session.""" + +import pytest +from evidence import FileTask +from terminal import AgentTerminal, ConfigureTerminal + +pytestmark = [pytest.mark.live, pytest.mark.main, pytest.mark.tui, pytest.mark.codex] + + +def test_ug_configure_codex_databricks(live_session, workspace): + """Scenario: configure Codex with Databricks Hosted and use its TUI. + + Expected: configure succeeds, Codex returns a file value through the real + gateway, exits normally, and can reopen the configuration ug created. + Optional AI Tools are disabled; the selected agent version is kept pinned. + """ + session = live_session + task = FileTask(session) + + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-upgrade", + "--disable-databricks-ai-tools", + timeout=240, + ) + assert not session.workspace_state().get("provider_services", {}).get("codex") + + with AgentTerminal(session, "codex", [str(session.binary), "codex"], "first-session") as tui: + tui.boot() + tui.submit(task.prompt) + tui.wait_for_task(task) + tui.exit_normally() + task.assert_completed(session, "codex") + session.assert_not_routed() + + with AgentTerminal(session, "codex", [str(session.binary), "codex"], "reopen") as tui: + tui.boot() + tui.check_input_and_exit() + + +def test_ug_configure_codex_openai_mps(live_session, workspace, codex_provider): + """Scenario: choose the real OpenAI MPS in ug configure's provider picker. + + Expected: ug saves that provider, and launching Codex without --provider + uses the saved choice to complete a file-reading task and exit normally. + """ + session = live_session + task = FileTask(session) + + command = [ + str(session.binary), + "configure", + "--workspaces", + workspace, + "--skip-upgrade", + "--disable-databricks-ai-tools", + ] + with ConfigureTerminal(session, "codex", command, "configure-provider") as configure: + configure.select_agent("Codex") + configure.choose("How should Codex get its models?", "External Models") + configure.choose("Select a model provider service:", codex_provider) + configure.finish(timeout=240) + assert session.workspace_state()["provider_services"]["codex"] == codex_provider + assert codex_provider in session.run("status").stdout + + with AgentTerminal(session, "codex", [str(session.binary), "codex"], "provider-session") as tui: + tui.boot() + tui.submit(task.prompt) + tui.wait_for_task(task) + tui.exit_normally() + task.assert_completed(session, "codex") + session.assert_not_routed() diff --git a/tests/test_integration_contract.py b/tests/test_integration_contract.py index 39c68d411..a637b9923 100644 --- a/tests/test_integration_contract.py +++ b/tests/test_integration_contract.py @@ -37,3 +37,18 @@ def test_integration_suite_uses_only_public_process_boundaries(): }: violations.append(f"{path.name}:{node.lineno}: uses {node.attr}") assert not violations, "\n".join(violations) + + +def test_integration_tests_describe_the_scenario_and_expected_result(): + root = Path(__file__).parent / "integration" + violations = [] + for path in root.rglob("test_*.py"): + for node in ast.walk(ast.parse(path.read_text())): + if not isinstance(node, ast.FunctionDef) or not node.name.startswith("test_"): + continue + description = ast.get_docstring(node) or "" + if "Scenario:" not in description or "Expected:" not in description: + violations.append(f"{path.name}:{node.lineno}: describe Scenario and Expected") + if any(arg.arg == "configured" for arg in node.args.args): + violations.append(f"{path.name}:{node.lineno}: setup must be visible in the test") + assert not violations, "\n".join(violations) From 6db29347b10081ab7103396db0990be84f1f3c3a Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Fri, 11 Sep 2026 16:12:26 -0400 Subject: [PATCH 3/6] Run the integration bootstrap with Python 3.12 in CI --- .github/workflows/integration.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index 02ee9a746..4900b8486 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -67,7 +67,7 @@ jobs: version: 0.9.8 - name: Test a fresh installed package without credentials run: | - python3 scripts/run_integration.py \ + uv run --no-project --python 3.12 python scripts/run_integration.py \ --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ --claude-version "$CLAUDE_VERSION" --codex-version "$CODEX_VERSION" \ --index-url "$PACKAGE_INDEX" --installation-only --output "$RUNNER_TEMP/ug-integration" @@ -139,7 +139,7 @@ jobs: if [[ "$DEPENDENCY" != unconstrained ]]; then args+=(--dependency "$DEPENDENCY") fi - python3 scripts/run_integration.py \ + uv run --no-project --python 3.12 python scripts/run_integration.py \ --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ --claude-version "$CLAUDE_VERSION" --codex-version "$CODEX_VERSION" \ --index-url "$PACKAGE_INDEX" --output "$RUNNER_TEMP/ug-integration" \ From 01b14f78ae77f7e7e62e66f0ac6e9eeb778ce898 Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Fri, 11 Sep 2026 16:28:50 -0400 Subject: [PATCH 4/6] Organize integration coverage as named top-level CUJs --- .github/workflows/integration.yml | 22 +- scripts/run_integration.py | 4 +- tests/AGENTS.md | 19 +- tests/README.md | 99 ++++---- tests/integration/README.md | 94 +++++--- tests/integration/conftest.py | 13 +- tests/integration/pytest.ini | 2 - .../test_ug_argument_forwarding.py | 77 ------ .../regressions/test_ug_headless_arguments.py | 129 ---------- .../regressions/test_ug_reconfigure.py | 81 ------- .../regressions/test_ug_tui_launch_modes.py | 46 ---- .../integration/test_smart_routing_claude.py | 46 +++- tests/integration/test_smart_routing_codex.py | 48 +++- tests/integration/test_ug_claude_commands.py | 55 +++++ tests/integration/test_ug_claude_headless.py | 225 ++++++++++++++++++ tests/integration/test_ug_codex_app_server.py | 34 +++ tests/integration/test_ug_codex_commands.py | 134 +++++++++++ tests/integration/test_ug_codex_headless.py | 129 ++++++++++ tests/integration/test_ug_configure_claude.py | 8 +- .../test_ug_configure_claude_lifecycle.py | 106 +++++++++ tests/integration/test_ug_configure_codex.py | 8 +- .../test_ug_configure_codex_lifecycle.py | 98 ++++++++ tests/integration/{ => utils}/Dockerfile | 2 +- .../{ => utils}/Dockerfile.dockerignore | 0 tests/integration/utils/__init__.py | 1 + tests/integration/{ => utils}/evidence.py | 92 ++++++- tests/integration/{ => utils}/harness.py | 13 - tests/integration/{ => utils}/terminal.py | 3 +- 28 files changed, 1111 insertions(+), 477 deletions(-) delete mode 100644 tests/integration/regressions/test_ug_argument_forwarding.py delete mode 100644 tests/integration/regressions/test_ug_headless_arguments.py delete mode 100644 tests/integration/regressions/test_ug_reconfigure.py delete mode 100644 tests/integration/regressions/test_ug_tui_launch_modes.py create mode 100644 tests/integration/test_ug_claude_commands.py create mode 100644 tests/integration/test_ug_claude_headless.py create mode 100644 tests/integration/test_ug_codex_app_server.py create mode 100644 tests/integration/test_ug_codex_commands.py create mode 100644 tests/integration/test_ug_codex_headless.py create mode 100644 tests/integration/test_ug_configure_claude_lifecycle.py create mode 100644 tests/integration/test_ug_configure_codex_lifecycle.py rename tests/integration/{ => utils}/Dockerfile (93%) rename tests/integration/{ => utils}/Dockerfile.dockerignore (100%) create mode 100644 tests/integration/utils/__init__.py rename tests/integration/{ => utils}/evidence.py (50%) rename tests/integration/{ => utils}/harness.py (96%) rename tests/integration/{ => utils}/terminal.py (99%) diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index 4900b8486..c381a0b43 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -11,8 +11,8 @@ on: suite: description: Which integration checks to run type: choice - options: [main, all, tui, regression, installation] - default: main + options: [live, tui, installation] + default: live ug_version: description: Exact ucode release, or checkout default: checkout @@ -30,6 +30,10 @@ on: description: Exact Codex version default: 0.154.0 required: true + dependency: + description: Optional exact dependency for reproduction (e.g. tomlkit==0.14.0) + default: '' + required: false index_url: description: Python package index containing the requested ug version default: https://pypi.org/simple @@ -108,19 +112,15 @@ jobs: PY live: - name: CUJs (${{ matrix.dependency }}) + name: CUJs needs: workspace runs-on: ubuntu-22.04 timeout-minutes: 45 - strategy: - fail-fast: false - matrix: - dependency: [unconstrained, tomlkit==0.14.0, tomlkit==0.15.1] env: UCODE_TEST_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} - DEPENDENCY: ${{ matrix.dependency }} - TEST_MARKER: ${{ inputs.suite == 'all' && 'live' || inputs.suite || 'main' }} + DEPENDENCY: ${{ inputs.dependency }} + TEST_MARKER: ${{ inputs.suite || 'live' }} steps: - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 with: @@ -136,7 +136,7 @@ jobs: shell: bash run: | args=() - if [[ "$DEPENDENCY" != unconstrained ]]; then + if [[ -n "$DEPENDENCY" ]]; then args+=(--dependency "$DEPENDENCY") fi uv run --no-project --python 3.12 python scripts/run_integration.py \ @@ -148,7 +148,7 @@ jobs: if: ${{ !cancelled() }} uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 with: - name: integration-live-${{ strategy.job-index }} + name: integration-cujs include-hidden-files: true path: | ${{ runner.temp }}/ug-integration/*.json diff --git a/scripts/run_integration.py b/scripts/run_integration.py index 1584e5ffd..a59c6b3d7 100644 --- a/scripts/run_integration.py +++ b/scripts/run_integration.py @@ -105,9 +105,9 @@ def arguments(): filters.add_argument("--maxfail", type=int) extra = args.pytest_args[1:] if args.pytest_args[:1] == ["--"] else args.pytest_args selected = filters.parse_args(extra) - marker = selected.m or ("installation" if args.installation_only else "main") + marker = selected.m or "live" if args.installation_only: - marker = f"installation and ({marker})" if marker else "installation" + marker = f"installation and ({selected.m})" if selected.m else "installation" args.pytest_args = [] for flag, value in (("-k", selected.k), ("-m", marker), ("--maxfail", selected.maxfail)): if value is not None: diff --git a/tests/AGENTS.md b/tests/AGENTS.md index 414123e2e..7d20b5dee 100644 --- a/tests/AGENTS.md +++ b/tests/AGENTS.md @@ -63,23 +63,26 @@ behavior a test claims to exercise. ### Required integration test format -- Organize the main suite around complete user journeys, with explicit names +- Organize the suite around complete user journeys, with explicit names such as `test_ug_configure_claude_databricks` or `test_smart_routing_codex_first_prompt`. Do not hide the agent/provider behind - generic parametrization in these main tests. + generic parametrization in these tests. - Every test has a docstring with **Scenario:** and **Expected:**. State what the user does and the observable evidence required for success, including limits. - Keep the configure command, launch, task, and assertions visible in the test. Fixtures provide fresh environments and credentials, never a preconfigured app. - Helpers may handle processes, terminal keys, transcript parsing, cleanup, and artifact collection. Do not bury an entire CUJ inside an opaque helper. -- Main configuration journeys must complete a real TUI task. A startup banner, +- Provider configuration journeys must complete a real TUI task. A startup banner, config file, echoed prompt, or tool output alone does not prove completion. -- Keep existing focused argument/lifecycle checks in `integration/regressions/`. - They remain runnable, with honest failures, separately from `-m main`. -- Current scope is the eight Claude/Codex provider and smart-routing CUJs. - Do not add MCP, skills, tracing, or the broad configure-option matrix without - a new scope request. +- Keep all CUJs as descriptive top-level `integration/test_*.py` files. Do not + create a separate regressions category. Shared process/terminal/evidence helpers + and Docker build files belong in `integration/utils/`; keep pytest entry points + and run documentation at the suite root. +- Current scope is basic Claude/Codex configuration, routing, script usage, + command forwarding, and configure/revert CUJs. Do not add MCP/skills functionality, + tracing, or the broad configure-option matrix without a new scope request. + Existing `mcp --help` checks cover dispatch only. - **Add:** state the user scenario and affected versions, choose the category, add a focused test and coverage row. Demonstrate regression failure on the diff --git a/tests/README.md b/tests/README.md index f459ed5d0..95b886fc9 100644 --- a/tests/README.md +++ b/tests/README.md @@ -1,70 +1,81 @@ # Test suites and user journeys -The integration suite runs a freshly installed ug wheel/release, exact real -Claude/Codex versions, and the existing real e2e workspace. It has no application -imports, mocks, monkeypatching, fake binaries/services, or fabricated ug state. +Integration runs a freshly installed ug wheel/release, exact real Claude/Codex +versions, and the existing real e2e workspace. It has no application imports, +mocks, monkeypatching, fake binaries/services, or fabricated ug state. | Category | Location | What it proves | | --- | --- | --- | | Unit/component | Existing `test_*.py` files | Individual behavior; dependencies may be mocked | | Existing e2e | `test_e2e*.py` | Real workspace behavior with some patched setup/internal calls | -| Main integration CUJs | `integration/test_ug_configure_*.py`, `integration/test_smart_routing_*.py` | Complete configure → real TUI → task → exit journeys | +| Integration CUJs | `integration/test_*.py` | Public configure, TUI, script, command, protocol, and lifecycle journeys | | Installation | `integration/test_installation.py` | Fresh installed package without credentials | -| Focused integration regressions | `integration/regressions/` | Public command, argument, protocol, and lifecycle contracts | -## Main CUJ matrix +## CUJ coverage matrix -These are **implemented assertions**, not a claim that every agent/version -combination passes. Consult the run's JUnit report and artifacts for results. -Each function states its **Scenario** and **Expected** outcome and shows its -configure and launch commands. No fixture silently configures the application. +These are **implemented assertions**, not a claim that every version passes. +Consult the run's JUnit report and artifacts for results. Each function states +its **Scenario** and **Expected** outcome and shows its configure and launch +commands. Fixtures supply fresh environments and credentials, never configured ug. +All tests live directly in `integration/`; shared mechanics live in `utils/`. -| Test | Setup and user action | Expected evidence | +| Test | User action | Expected evidence | | --- | --- | --- | -| `test_ug_configure_claude_databricks` | Configure Claude with Databricks Hosted; open TUI and read a file | Assistant returns an unpredictable file value, exits normally, and reopens with working keyboard input | -| `test_ug_configure_claude_anthropic_mps` | Select the existing Anthropic MPS in the real configure picker; launch without a provider override | Saved provider appears in status; real Claude completes the file task and exits | -| `test_ug_configure_codex_databricks` | Configure Codex with Databricks Hosted; open TUI and read a file | Completed assistant answer contains the file value, normal exit and reopen | -| `test_ug_configure_codex_openai_mps` | Select the existing OpenAI MPS in the real configure picker; launch without a provider override | Saved provider appears in status; real Codex completes the file task and exits | -| `test_smart_routing_claude_first_prompt` | Configure, enable routing, type the first TUI prompt | Real routing decision and prompt replay plus completed task; no routing before submission | -| `test_smart_routing_codex_first_prompt` | Configure, enable routing, type the first TUI prompt | Real routing decision plus completed task; no fallback or routing before submission | -| `test_smart_routing_claude_subagent` | Ask Claude to delegate a file-reading task | Actual child transcript with the answer, correlated routing decision/child start, parent answer | -| `test_smart_routing_codex_subagent` | Ask Codex to delegate a file-reading task | Actual child session with the answer, correlated routing decision/child start, parent answer | +| `test_ug_configure_claude_databricks` | Configure Databricks Hosted; open Claude TUI and read a file | Assistant returns an unpredictable file value; normal exit; reopen with working keyboard input | +| `test_ug_configure_claude_anthropic_mps` | Select Anthropic MPS in the real configure picker; launch Claude | Saved provider in status; completed TUI file task; normal exit | +| `test_ug_configure_codex_databricks` | Configure Databricks Hosted; open Codex TUI and read a file | Completed assistant answer contains the file value; normal exit and reopen | +| `test_ug_configure_codex_openai_mps` | Select OpenAI MPS in the real configure picker; launch Codex | Saved provider in status; completed TUI file task; normal exit | +| `test_smart_routing_claude_first_prompt` | Enable routing and type the first TUI prompt | Real routing decision and replay; completed task; no routing before submission; reopen | +| `test_smart_routing_codex_first_prompt` | Enable routing and type the first TUI prompt | Real routing decision; completed task; no fallback or routing before submission; reopen | +| `test_smart_routing_claude_subagent` | Delegate a file-reading task | Child answer, correlated routing decision/child start, parent answer | +| `test_smart_routing_codex_subagent` | Delegate a file-reading task | Native child-parent linkage; completed child turn uses the routed model; parent answer | +| `test_smart_routing_claude_explicit_model_bypasses_routing` | Launch TUI with an explicit model and routing enabled | Completed task, no routing wrapper, normal exit and reopen | +| `test_smart_routing_codex_explicit_model_bypasses_routing` | Launch TUI with an explicit model and routing enabled | Completed task, no routing wrapper, normal exit and reopen | +| `test_ug_claude_headless_prompt_argument`, `test_ug_claude_headless_prompt_stdin`, `test_ug_claude_headless_prompt_after_separator` | Run Claude from a script using each prompt form | Structured final answer contains the file value; exit zero; no routing | +| `test_ug_codex_headless_prompt_argument`, `test_ug_codex_headless_prompt_stdin`, `test_ug_codex_headless_prompt_after_separator` | Run Codex from a script using each prompt form | Completed turn and final answer contain the file value; exit zero; no routing | +| `test_ug_claude_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` with routing enabled | Real file task completes; no routing wrapper | +| `test_ug_codex_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` / `-m VALUE` with routing enabled | Real file task completes; no routing wrapper | +| `test_ug_claude_preserves_caller_settings_and_hook` | Pass a settings path containing spaces | Real SessionStart hook executes; caller file unchanged; file task completes | +| `test_ug_claude_reports_unsupported_short_model_option` | Pass Claude's unsupported `-m` | Actual agent error and exit status preserved | +| `test_ug_claude_auth_help`, `test_ug_claude_mcp_help` | Request subcommand help, routing off/on | Real agent help; no routing wrapper | +| `test_ug_codex_app_help`, `test_ug_codex_app_server_help`, `test_ug_codex_exec_help`, `test_ug_codex_mcp_help` | Request subcommand help, routing off/on | Real agent help; no routing wrapper | +| `test_ug_codex_app_reports_unknown_argument` | Pass an invalid option directly to `ug codex app`, routing off/on | Real Codex parser error and status preserved | +| `test_ug_codex_app_server_client_initializes` | Connect a stdio client, direct/`--` separator, routing off/on | Actual JSON-RPC initialize response; no non-JSON stdout; no routing | +| `test_ug_configure_claude_repeat_and_revert`, `test_ug_configure_codex_repeat_and_revert` | Configure twice over user settings; complete a task; revert twice | Settings preserved; no bearer in ug state; generated config removed; status unconfigured | +| `test_ug_configure_claude_rejects_invalid_credentials`, `test_ug_configure_codex_rejects_invalid_credentials` | Configure with a rejected bearer against the real workspace | Authentication failure; no successful saved setup | +| `test_ug_installed_wheel_exposes_help_and_version` | Invoke freshly installed console command | Package version matches; public help works | +| `test_ug_status_in_fresh_home_is_unconfigured` | Request status before configure | Unconfigured status | +| `test_ug_auth_without_configuration_explains_how_to_configure` | Request auth before configure | Actionable setup error and nonzero exit | -The main configuration tests include ug's normal validation. Routing tests use -`--skip-validate` during setup because their own TUI task is the validation. -All keep the requested agent versions with `--skip-upgrade` and disable optional -Databricks AI Tools to keep these basic journeys focused. +With both agents selected there are **45 live cases** (10 interactive TUI cases) +and **3 installation checks**. Parametrization varies argument spelling or routing +mode, never hides the agent/provider in the test name. Duplicate boot-only cases +were merged into the Databricks, first-prompt, and explicit-model TUI journeys. +Generated-file cleanup and strict app-server stdout assertions remain enforced. -## Retained regression coverage +Provider configuration tests include ug's normal validation. Other setup uses +`--skip-validate` when the journey supplies its own task or checks a command +contract. Tests use `--skip-upgrade` to preserve selected agent versions and disable +optional Databricks AI Tools. Help forwarding does not claim MCP functionality. -The original focused checks moved into `integration/regressions/`; they were not -deleted when the main suite was narrowed. Select them with `-- -m regression`. - -| Concern | Coverage | -| --- | --- | -| Fresh consumer resolution differs from `uv.lock` (#496) | Fresh wheel install; dependency constraints/replay; CI tests unconstrained, tomlkit 0.14.0 and 0.15.1 | -| Reconfigure, preserve user settings, revert | CLI-created state and real files; known generated-file cleanup failure remains an assertion | -| Rejected credentials | Real workspace rejection and no successful saved setup | -| Subcommand forwarding (#502) | Real Claude auth/mcp and Codex app/app-server/exec/mcp help, routing off/on | -| Codex app argument error | Real parser error and exit status; excludes unrelated per-launch warnings | -| Codex app-server | Real JSON-RPC initialize, direct/separator forms; non-JSON stdout still fails | -| Headless prompt/model arguments | Real file task, stdin/separators and caller settings/hook; Claude's unsupported `-m` must retain its real error | -| TUI boot modes | Routing off/on/explicit-model, first boot/reopen, input/clear/exit; no inference claim | +Fresh consumer dependency resolution covers the install path behind #496, rather +than consuming `uv.lock`. Use `--dependency PACKAGE==VERSION` or replay the archived +dependency graph to reproduce a user's combination. PR CI runs one live CUJ job. ## Gaps and deferred scope | Scenario | Status / requirement | | --- | --- | -| MCP and skills CUJs | Deferred at the user's request | -| Broad configure flags, tracing, multiple workspaces, OAuth/PAT flows | Deferred while focusing on basic main CUJs | -| Provider switching, relayed/subscription MPS | Not covered by the four basic provider journeys | -| Initial prompt supplied on the launch command line | Not yet covered by main routing CUJs | +| MCP and skills functionality | Deferred at the user's request; existing `mcp --help` dispatch checks only | +| Broad configure flags, tracing, multiple workspaces, OAuth/PAT flows | Deferred while focusing on basic CUJs | +| Provider switching, relayed/subscription MPS | Not covered by the four provider journeys | +| TUI initial prompt supplied on the launch command line | Not yet covered; headless prompt arguments are covered | | Follow-up turns and conversation resume | Not covered; reopen proves startup, not conversation resume | -| Exact child model identity | Verified only when the real agent reports it; missing model fields remain unknown in artifacts | -| Full allow/deny tool-permission matrix | Not covered; real onboarding/trust choices are handled through the TUI | +| Claude child model identity | Verified only when the actual start event reports it; missing fields remain unknown | +| Full allow/deny tool-permission matrix | Not covered; onboarding/trust uses actual TUI choices | | Desktop Codex app, Isaac itself, auto-upgrades | Not covered by command forwarding or pinned-version tests | | Native macOS/Windows managed settings, resize/signals | Separate platform coverage needed | -| Other agents | Current main scope is Claude Code and Codex | +| Other agents | Current scope is Claude Code and Codex | See [integration/README.md](integration/README.md) for commands, CI, artifacts, and reproduction. Follow [AGENTS.md](AGENTS.md) and [CLAUDE.md](CLAUDE.md) when diff --git a/tests/integration/README.md b/tests/integration/README.md index c18f25462..cf0c2851e 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -22,7 +22,7 @@ in that case; the runner never edits or bypasses those managed settings. Use the existing e2e workspace and its `DATABRICKS_BEARER` credential. Locally, `--profile YOUR_PROFILE` can mint a bearer for an explicitly selected profile. -No profile or workspace is selected automatically. The default test selection is `main`. +No profile or workspace is selected automatically. The default test selection is `live` (all live CUJs). ```bash export UCODE_TEST_WORKSPACE=https://your-existing-e2e-workspace @@ -43,7 +43,7 @@ for an npm mirror if public npm is unavailable. Older releases that only provide the `ucode` command require `--entry-point ucode`. Select one agent by providing only its version. Exact agent versions are -required; floating `latest`, caret, and tilde versions are rejected. Main CUJs use the workspace's configuration and need no model input. Regression cases that +required; floating `latest`, caret, and tilde versions are rejected. Provider and default-routing CUJs use the workspace's configuration and need no model input. Cases that exercise explicit model arguments use a real `system.ai` model already discovered by `ug configure`, recorded in that case's `model.json`. Optional `--claude-model` and `--codex-model` overrides reproduce a particular model-related failure. @@ -54,7 +54,7 @@ python3 scripts/run_integration.py \ --ug-version checkout --claude-version 2.1.268 --codex-version 0.154.0 \ --claude-model YOUR_CLAUDE_MODEL --codex-model YOUR_CODEX_MODEL \ --profile YOUR_PROFILE --dependency tomlkit==0.14.0 \ - -- -k 'app_server or subcommand' + -- -k 'app_server or app_help' ``` Repeat with `--dependency tomlkit==0.15.1`, or run without constraints to test @@ -73,23 +73,35 @@ integration pass. Requested live checks fail when credentials, binaries, models, or capabilities are missing. There are no capability-based skips or retries of failed model tasks. A failing historical version should remain a failing result. -## Main tests and format +## Test layout and format -The default selection is the **eight main end-to-end CUJs** in: +All user journeys are top-level tests. There is no separate regressions category: ```text -test_ug_configure_claude.py # Databricks Hosted and Anthropic MPS -test_ug_configure_codex.py # Databricks Hosted and OpenAI MPS -test_smart_routing_claude.py # first prompt and real subagent -test_smart_routing_codex.py # first prompt and real subagent +test_ug_configure_claude.py # Databricks Hosted and Anthropic MPS +test_ug_configure_codex.py # Databricks Hosted and OpenAI MPS +test_smart_routing_claude.py # first prompt, subagent, explicit model +test_smart_routing_codex.py # first prompt, subagent, explicit model +test_ug_claude_headless.py # script prompts, models, caller settings +test_ug_codex_headless.py # script prompts and model arguments +test_ug_claude_commands.py # command help forwarding +test_ug_codex_commands.py # command help and parser error forwarding +test_ug_codex_app_server.py # actual client/server initialize exchange +test_ug_configure_claude_lifecycle.py # repeat setup, revert, rejected credentials +test_ug_configure_codex_lifecycle.py # repeat setup, revert, rejected credentials +test_installation.py # fresh installed package +utils/ # process/terminal/evidence helpers and Docker files ``` +`conftest.py`, `pytest.ini`, and this README stay at the suite root for pytest +discovery and run instructions. + Each test has a `Scenario:` / `Expected:` docstring and shows its own public -configure command, TUI launch, user task, and assertions. Shared code only handles +configure command, launch, user action, and assertions. Shared code only handles process/terminal mechanics, evidence, and cleanup. Fixtures supply an isolated session and credentials; none manufacture or configure application state. -Configuration CUJs use normal ug validation, then require their own completed +Provider CUJs use normal ug validation, then require their own completed interactive task. Routing CUJs skip the preliminary validation prompt and require the actual routed TUI task instead. Tests disable optional Databricks AI Tools and pass `--skip-upgrade` to preserve the selected version. They use real onboarding @@ -99,8 +111,9 @@ A fixture file contains an unpredictable value absent from the prompt. Success requires an assistant answer in the real agent transcript containing that value, plus normal TUI exit. Codex evidence requires its task-complete event. Subagent CUJs require a separate child transcript, child answer, and a correlated routing -decision/start event. If an agent omits its child's model, the artifact records -that unknown; the test does not claim exact child model verification. +decision. Claude uses its real child-start audit; its model is unknown when the +event omits it. Codex requires native parent linkage and a completed child turn +using the routed model, excluding inherited parent turns. MPS CUJs select the existing services already used by e2e: @@ -111,23 +124,24 @@ Use `--claude-provider` / `--codex-provider` to reproduce another existing servi Those names are recorded in `versions.json`. No service is created or modified. A missing service or permission fails the selected CUJ, rather than skipping it. -There are 49 cases with both agents: 8 main CUJs, 3 installation checks, and -38 retained regressions. Choose deliberately: +There are **45 live cases** (including 10 TUI journeys) and **3 installation +checks** with both agents. See the named coverage and gaps matrix in +[../README.md](../README.md). ```bash # Append one of these selections to the runner command: --- -m main # default: eight complete user journeys --- -m installation # package checks (use --installation-only to need no auth) --- -m regression # retained argument, lifecycle, and boot checks --- -m live # main CUJs plus all live regressions --- -m tui # main CUJs plus focused boot regressions +-- -m live # default: all live user journeys +-- -m tui # ten complete interactive TUI journeys +-- -k test_ug_codex_app_server_client_initializes # one named journey and its variants +# Use --installation-only before -- for package checks without credentials. ``` -The prior generic tests moved under `regressions/` to keep the main journeys easy -to read. Real failures, including generated config left after revert and banners -on app-server stdout, remain assertions in those regressions. The coverage and -gaps matrix is in [../README.md](../README.md). MCP, skills, tracing, the broad -configure-option matrix, and other agents are outside this focused revision. +The old focused checks are now descriptive CUJs with setup and outcomes visible +in each test. Duplicate boot-only checks are incorporated into the Databricks, +first-prompt, and explicit-model TUI journeys. Real failures, including generated +config left after revert and banners on app-server stdout, remain assertions. +MCP/skills functionality, tracing, the broad configure-option matrix, and other +agents are outside this focused revision. ## Reproduce a failure @@ -145,7 +159,7 @@ Each run writes a new `.integration-runs//` directory containing: - `wheels/`: the tested wheel when built from the checkout; replay it with `--ug-wheel`. For release installations, `installed.txt` records the resolution. -Teardown invokes real `ug revert` when setup created state, restoring machine-level +Teardown invokes real `ug revert` through a PTY when setup created state, restoring machine-level configuration through the public CLI. Per-test homes and working directories are then deleted even on failure. The working directory is outside the checkout so an agent cannot inherit its @@ -162,7 +176,10 @@ selection to installation checks, including when additional filters are used. ## Run in GitHub Actions The **Integration** workflow runs on relevant pull requests and pushes to `main`. -Its installation job needs no credentials. For same-repository PRs, the live jobs +It runs directly on fresh GitHub Ubuntu VMs, not inside the optional Docker image. +Local native runs use the same runner; Colima/Docker provides a separate Linux +container option. Matching dependency versions does not make those OS environments identical. +Its installation job needs no credentials. For same-repository PRs, the live job reuse the existing `UCODE_TEST_WORKSPACE` and `DATABRICKS_BEARER` secrets. Fork PRs run installation checks only because they cannot receive those secrets. @@ -172,19 +189,20 @@ is accepted). It never changes the secret or switches workspaces. There is no CI model-discovery or model-selection job. Real `ug configure` performs its normal workspace discovery inside each test; only explicit-model scenarios choose and record a discovered `system.ai` model as a test argument. -PR CI runs the eight main CUJs in each dependency job; `all` explicitly includes -the retained regressions. The live matrix covers unconstrained resolution, -tomlkit 0.14.0, and tomlkit 0.15.1. Jobs use Ubuntu 22.04; newer Ubuntu runner +PR CI runs all 45 live cases in one **CUJs** job with fresh consumer dependency +resolution. There is no default dependency matrix. Manual dispatch accepts an +optional `dependency` such as `tomlkit==0.14.0`, equivalent to the local runner's +`--dependency` option. Jobs use Ubuntu 22.04; newer Ubuntu runner policies prevented Codex's bubblewrap tool from reading even the test file in the first run. The agent sandbox is not disabled or bypassed. The workflow consumes the stored bearer; it does not mint or refresh credentials. For a manual run, use **Actions → Integration → Run workflow**, select the branch, -and choose `main` (default), `all`, `tui`, `regression`, or `installation`. Set the ug/agent versions. From the CLI: +and choose `live` (default), `tui`, or `installation`. Set the ug/agent versions. From the CLI: ```bash gh workflow run integration.yml -R databricks/unity-gateway --ref YOUR_BRANCH \ - -f suite=main -f ug_version=checkout \ + -f suite=live -f ug_version=checkout \ -f claude_version=2.1.268 -f codex_version=0.154.0 gh run list -R databricks/unity-gateway --workflow integration.yml gh run watch RUN_ID -R databricks/unity-gateway --exit-status @@ -204,11 +222,11 @@ the CI workspace. Select that local profile explicitly; CI secrets are not downl ```bash gh run download RUN_ID -R databricks/unity-gateway \ - -n integration-live-0 -D .integration-runs/from-ci + -n integration-cujs -D .integration-runs/from-ci ``` -Use `integration-live-1` or `integration-live-2` for the other dependency jobs, -or `integration-installation` for package failures. Read `versions.json` for the +Use `integration-installation` for package failures. Older runs used numbered +`integration-live-*` artifacts; download the name shown on that run. Read `versions.json` for the exact agent versions, model overrides, entry point, platform and source revision. For an explicit-model case without a runner override, read its `model.json` for the exact model used. Basic boot cases require no model arguments. Use @@ -234,7 +252,7 @@ use the same OS/architecture as the original run; add `--platform linux/amd64` to both `docker build` and `docker run` on an ARM Mac to match GitHub's Ubuntu runner. Changing platforms or resolving a fresh npm lock is a new comparison, not an exact dependency replay. -Use `-- -m main` for the eight CUJs or +Use `-- -m tui` for the ten interactive journeys or `-- -k test_smart_routing_codex_first_prompt` to narrow a failure. Each rerun needs a new output directory. Inspect: - `junit.xml` for the failing case and assertion. @@ -260,7 +278,7 @@ versions inside it. Build from the repository root: colima start COPYFILE_DISABLE=1 tar --format=ustar --exclude=__pycache__ --exclude=.pytest_cache \ -cf - scripts/run_integration.py tests/integration | \ - docker build -f tests/integration/Dockerfile -t ug-integration - + docker build -f tests/integration/utils/Dockerfile -t ug-integration - # Reuse the same e2e variables. Credentials are passed at runtime, never built # into the image. The named volume keeps results after the container exits. @@ -271,7 +289,7 @@ docker run --rm --init \ ug-integration \ --ug-version YOUR_RELEASE_VERSION \ --claude-version 2.1.268 --codex-version 0.154.0 \ - -- -m main + -- -m live ``` Use a new results volume for each run, or pass a new `--output /results/NAME`. diff --git a/tests/integration/conftest.py b/tests/integration/conftest.py index 4930430f6..f3ce72b29 100644 --- a/tests/integration/conftest.py +++ b/tests/integration/conftest.py @@ -9,7 +9,8 @@ from pathlib import Path import pytest -from harness import UserSession +from utils.harness import UserSession +from utils.terminal import TerminalProcess def pytest_collection_modifyitems(config, items): @@ -63,8 +64,14 @@ def session(request, installed_binary): finally: # Restore machine-level settings through the same public CLI that # created them. A later fresh-home test must not inherit this setup. - if (user.home / ".ucode/state.json").is_file(): - user.run("revert") + if any( + (user.home / ".ucode" / name).is_file() + for name in ("state.json", "managed-backups/manifest.json") + ): + with TerminalProcess( + user, "ug", [str(user.binary), "revert"], "cleanup-revert" + ) as terminal: + terminal.finish() @pytest.fixture diff --git a/tests/integration/pytest.ini b/tests/integration/pytest.ini index 577c7538d..ed06ccb80 100644 --- a/tests/integration/pytest.ini +++ b/tests/integration/pytest.ini @@ -2,8 +2,6 @@ testpaths = . addopts = --strict-markers --tb=short markers = - main: complete configuration and smart-routing user journeys - regression: focused command and lifecycle regression checks installation: installed-package checks that require no workspace live: requires the real workspace used by the existing e2e suite tui: real interactive terminal boot, keyboard input, exit and reopen diff --git a/tests/integration/regressions/test_ug_argument_forwarding.py b/tests/integration/regressions/test_ug_argument_forwarding.py deleted file mode 100644 index d98c91e67..000000000 --- a/tests/integration/regressions/test_ug_argument_forwarding.py +++ /dev/null @@ -1,77 +0,0 @@ -"""Regression cases for #502 and real Codex config serialization (#496).""" - -import pytest - -pytestmark = [pytest.mark.live, pytest.mark.regression] - -# Compare against each real agent's output. A wrapper's own help is not evidence -# that it dispatched the requested subcommand. No updater or desktop app is run. -HELP_CASES = [ - pytest.param("codex", ["app", "--help"], marks=pytest.mark.codex, id="codex-app"), - pytest.param("codex", ["app-server", "--help"], marks=pytest.mark.codex, id="codex-app-server"), - pytest.param("codex", ["exec", "--help"], marks=pytest.mark.codex, id="codex-exec"), - pytest.param("codex", ["mcp", "--help"], marks=pytest.mark.codex, id="codex-mcp"), - pytest.param("claude", ["mcp", "--help"], marks=pytest.mark.claude, id="claude-mcp"), - pytest.param("claude", ["auth", "--help"], marks=pytest.mark.claude, id="claude-auth"), -] - - -@pytest.mark.parametrize("agent,args", HELP_CASES) -@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_ug_subcommand_help_reaches_real_agent(live_session, workspace, agent, args, routing): - """Scenario: request agent subcommand help through ug with routing off/on. - - Expected: the real agent's help is returned and no routing wrapper starts. - """ - session = live_session - session.configure(agent, workspace) - session.env["ENABLE_SMART_ROUTING_V2"] = routing - expected = session.run(*args, binary=agent).stdout.strip() - assert expected - actual = session.run(agent, "--", *args).stdout - assert expected in actual, actual - session.assert_not_routed() - - -@pytest.mark.codex -@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_ug_codex_app_preserves_unknown_argument_error(live_session, workspace, routing): - """Scenario: pass an unknown option to ug codex app. - - Expected: the real Codex parser's error and exit code survive forwarding. - Per-launch helper warnings are not part of the argument-error contract. - """ - session = live_session - args = ["app", "--ug-integration-unknown-option"] - session.configure("codex", workspace) - expected = session.run(*args, binary="codex", ok=False) - assert expected.returncode != 0 and "error:" in expected.stderr - session.env["ENABLE_SMART_ROUTING_V2"] = routing - # Exercise the direct subcommand form without opening a desktop app or - # allowing ug's own --help option to intercept the request. - actual = session.run("codex", *args, ok=False) - assert actual.returncode == expected.returncode - parser_error = expected.stderr[expected.stderr.index("error:") :].strip() - assert parser_error in actual.stderr, actual.stdout + actual.stderr - session.assert_not_routed() - - -@pytest.mark.codex -@pytest.mark.parametrize("separator", [False, True], ids=["direct", "launcher-separator"]) -@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) -def test_ug_codex_app_server_initializes_over_clean_json_rpc( - live_session, workspace, separator, routing -): - """Scenario: connect a real client to ug codex app-server over stdio. - - Expected: initialize returns a valid response on stdout with no non-JSON - banners mixed into the protocol; direct and launcher separator forms work. - """ - session = live_session - session.configure("codex", workspace) - session.env["ENABLE_SMART_ROUTING_V2"] = routing - args = ["app-server", "--listen", "stdio://"] - if separator: - args.insert(0, "--") - session.app_server_handshake(args) - session.assert_not_routed() diff --git a/tests/integration/regressions/test_ug_headless_arguments.py b/tests/integration/regressions/test_ug_headless_arguments.py deleted file mode 100644 index 7d8f8c977..000000000 --- a/tests/integration/regressions/test_ug_headless_arguments.py +++ /dev/null @@ -1,129 +0,0 @@ -"""Real file-reading tasks, including launcher-style options after `--`.""" - -import json -import uuid - -import pytest - -pytestmark = [pytest.mark.live, pytest.mark.regression] - - -@pytest.mark.parametrize( - "agent", - [ - pytest.param("claude", marks=pytest.mark.claude), - pytest.param("codex", marks=pytest.mark.codex), - ], -) -@pytest.mark.parametrize( - "model_form,prompt_form", - [ - pytest.param("separate", "argument", id="model-value"), - pytest.param("equals", "argument", id="model-equals"), - pytest.param("short", "argument", id="model-short"), - pytest.param("separate", "stdin", id="stdin-prompt"), - pytest.param("separate", "separator", id="nested-separator"), - ], -) -def test_ug_headless_task_accepts_model_and_prompt_arguments( - live_session, workspace, agent, model_form, prompt_form -): - """Scenario: forward explicit model flags and argument/stdin prompts through ug. - - Expected: a real headless agent returns an unknown file value; Claude also - runs the caller's hook. Unsupported options preserve the real agent error. - """ - session = live_session - session.configure(agent, workspace) - model = session.model_for_explicit_case(agent) - nonce = uuid.uuid4().hex - (session.cwd / "input.txt").write_text(nonce + "\n") - prompt = "Read input.txt in the current directory using a tool. Reply with only its contents." - # An explicit model must bypass routing even when globally enabled. - session.env["ENABLE_SMART_ROUTING_V2"] = "1" - model_args = { - "separate": ["--model", model], - "equals": [f"--model={model}"], - "short": ["-m", model], - }[model_form] - input_text = prompt + "\n" if prompt_form == "stdin" else None - # Both separators are real: the outer one belongs to ug, the inner one - # belongs to the selected agent. The prompt remains one argument. - prompt_args = ["--", prompt] if prompt_form == "separator" else [prompt] - if agent == "claude": - # A real caller-supplied settings file exercises the merge needed by - # launchers such as Isaac. Its hook must execute without losing ug auth. - caller_settings = session.cwd / "caller settings.json" - caller_settings.write_text( - json.dumps( - { - "hooks": { - "SessionStart": [ - { - "hooks": [ - { - "type": "command", - "command": "echo caller-hook-ran > caller-hook.txt", - } - ] - } - ] - }, - } - ) - ) - args = [ - "-p", - *model_args, - "--max-turns", - "4", - "--output-format", - "json", - "--allowedTools", - "Read", - "--settings", - str(caller_settings), - *(prompt_args if input_text is None else []), - ] - else: - args = [ - "exec", - "--skip-git-repo-check", - "--json", - *model_args, - *(prompt_args if input_text is None else ["-"]), - ] - if agent == "claude" and model_form == "short": - # Claude 2.1.268 has no -m option. Verify forwarding against the actual - # binary instead of inventing support that the upstream CLI lacks. - expected = session.run(*args, binary=agent, ok=False, input_text=input_text) - actual = session.run(agent, "--", *args, ok=False, input_text=input_text) - assert expected.returncode != 0 and "unknown option '-m'" in expected.stderr - assert actual.returncode == expected.returncode - assert "unknown option '-m'" in actual.stderr - session.assert_not_routed() - return - result = session.run(agent, "--", *args, timeout=180, input_text=input_text) - # Look in the agent's structured final output, never in the echoed prompt, - # a tool request, or a banner. The nonce was not given to the model. - payloads = [] - for line in result.stdout.splitlines(): - try: - payloads.append(json.loads(line)) - except ValueError: - pass - if agent == "claude": - final = [p for p in payloads if isinstance(p, dict) and p.get("type") == "result"] - assert final and not final[-1].get("is_error"), result.stdout - assert nonce in final[-1].get("result", ""), result.stdout - assert (session.cwd / "caller-hook.txt").read_text().strip() == "caller-hook-ran" - else: - final = [ - p["item"].get("text", "") - for p in payloads - if isinstance(p, dict) - and p.get("type") == "item.completed" - and p.get("item", {}).get("type") == "agent_message" - ] - assert any(nonce in text for text in final), result.stdout - session.assert_not_routed() diff --git a/tests/integration/regressions/test_ug_reconfigure.py b/tests/integration/regressions/test_ug_reconfigure.py deleted file mode 100644 index 08db3214b..000000000 --- a/tests/integration/regressions/test_ug_reconfigure.py +++ /dev/null @@ -1,81 +0,0 @@ -"""Configure, repeat, and undo real setup; never manufacture ug state.""" - -import json -import tomllib - -import pytest - -pytestmark = [pytest.mark.live, pytest.mark.regression] - - -@pytest.mark.parametrize( - "agent", - [ - pytest.param("claude", marks=pytest.mark.claude), - pytest.param("codex", marks=pytest.mark.codex), - ], -) -def test_ug_reconfigure_and_revert_preserve_user_settings(live_session, workspace, agent): - """Scenario: configure twice over existing user settings, then revert twice. - - Expected: unrelated settings survive and ug's generated config is removed. - This is a lifecycle check; the main CUJs separately prove task completion. - """ - session = live_session - if agent == "claude": - user_path = session.home / ".claude/settings.json" - user_settings = '{"permissions":{"allow":["Read"]},"env":{"UG_USER_SETTING":"keep"}}\n' - load = json.loads - else: - user_path = session.home / ".codex/config.toml" - user_settings = "# user comment\n[notice]\nhide_rate_limit_model_nudge = true\n" - load = tomllib.loads - user_path.parent.mkdir(parents=True) - user_path.write_text(user_settings) - - session.configure(agent, workspace) - first = session.state() - ws_state = first["workspaces"][workspace] - assert first["current_workspace"] == workspace - assert agent in ws_state["available_tools"] - assert ws_state[f"{agent}_models"], "Discovery returned no models for the selected agent" - assert workspace in session.run("status").stdout - session.configure(agent, workspace) - second = session.state() - assert second["current_workspace"] == workspace - assert second["workspaces"][workspace]["available_tools"] == ws_state["available_tools"] - current = load(user_path.read_text()) - assert all(current.get(key) == value for key, value in load(user_settings).items()) - for path in (session.home / ".ucode").rglob("*"): - if path.is_file(): - assert session.env["DATABRICKS_BEARER"] not in path.read_text(errors="replace"), ( - f"Bearer was persisted in {path.name}" - ) - - session.run("revert") - assert load(user_path.read_text()) == load(user_settings) - assert "Not Configured" in session.run("status").stdout - private = ".claude/ucode-settings.json" if agent == "claude" else ".codex/ucode.config.toml" - assert not (session.home / private).exists() - session.run("revert") # Reverting an already reverted setup is safe. - - -@pytest.mark.parametrize( - "agent", - [ - pytest.param("claude", marks=pytest.mark.claude), - pytest.param("codex", marks=pytest.mark.codex), - ], -) -def test_ug_configure_rejects_invalid_workspace_credentials(live_session, workspace, agent): - """Scenario: configure against the real workspace with an invalid bearer. - - Expected: the service rejects authentication and ug saves no successful setup. - """ - live_session.env["DATABRICKS_BEARER"] = "ug-integration-intentionally-invalid" - result = live_session.configure(agent, workspace, ok=False) - assert result.returncode != 0 - output = result.stdout + result.stderr - assert "rejected the access token" in output or "401" in output, output - assert "Configuration Complete" not in output - assert not (live_session.home / ".ucode/state.json").exists() diff --git a/tests/integration/regressions/test_ug_tui_launch_modes.py b/tests/integration/regressions/test_ug_tui_launch_modes.py deleted file mode 100644 index d5aaf9f7d..000000000 --- a/tests/integration/regressions/test_ug_tui_launch_modes.py +++ /dev/null @@ -1,46 +0,0 @@ -"""Boot the real interactive terminal, then reopen the state it actually wrote.""" - -import pytest -from terminal import AgentTerminal - -pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.regression] - - -@pytest.mark.parametrize("launch", ["routing-off", "routing-on", "explicit-model"]) -@pytest.mark.parametrize( - "agent", - [ - pytest.param("claude", marks=pytest.mark.claude), - pytest.param("codex", marks=pytest.mark.codex), - ], -) -def test_ug_tui_launch_mode_supports_boot_reopen_and_exit(live_session, workspace, agent, launch): - """Scenario: boot and reopen a configured agent in each routing launch mode. - - Expected: onboarding, keyboard editing, and normal exit work; wrappers start - only when requested. This regression makes no inference-completion claim. - """ - session = live_session - session.configure(agent, workspace) - session.env["ENABLE_SMART_ROUTING_V2"] = "0" if launch == "routing-off" else "1" - args = [] - if launch == "explicit-model": - model = session.model_for_explicit_case(agent) - args = ["--", "--model", model] - - for name in ("first-boot", "reopen"): - command = [str(session.binary), agent, *args] - with AgentTerminal(session, agent, command, name) as terminal: - terminal.boot() - terminal.check_input_and_exit() - - if launch == "routing-on": - filename = "claude-v2-pty.log" if agent == "claude" else "codex-v2-interposer.log" - path = session.home / ".ucode" / filename - assert path.is_file(), "The real routing wrapper did not start" - log = path.read_text() - session.record("routing-boot.json", {"log": log}) - assert "[READY]" in log, log - assert "[ROUTE]" not in log, "Booting without a model prompt should not route a turn" - else: - session.assert_not_routed() diff --git a/tests/integration/test_smart_routing_claude.py b/tests/integration/test_smart_routing_claude.py index ac6ced5e7..1834760e1 100644 --- a/tests/integration/test_smart_routing_claude.py +++ b/tests/integration/test_smart_routing_claude.py @@ -1,10 +1,10 @@ -"""Main CUJs: real first-prompt and child-task routing in Claude's TUI.""" +"""CUJs: real first-prompt and child-task routing in Claude's TUI.""" import pytest -from evidence import FileTask, assert_subagent_routed -from terminal import AgentTerminal +from utils.evidence import FileTask, assert_subagent_routed +from utils.terminal import AgentTerminal -pytestmark = [pytest.mark.live, pytest.mark.main, pytest.mark.tui, pytest.mark.claude] +pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.claude] def test_smart_routing_claude_first_prompt(live_session, workspace): @@ -37,6 +37,9 @@ def test_smart_routing_claude_first_prompt(live_session, workspace): assert "[REPLAY] first prompt submitted" in log, log tui.exit_normally() task.assert_completed(session, "claude") + with AgentTerminal(session, "claude", command, "reopen") as tui: + tui.boot() + tui.check_input_and_exit() def test_smart_routing_claude_subagent(live_session, workspace): @@ -65,6 +68,39 @@ def test_smart_routing_claude_subagent(live_session, workspace): tui.submit(task.delegate_prompt) tui.wait_for_task(task, timeout=240) task.assert_completed(session, "claude", child=True) - assert_subagent_routed(session, "claude") + assert_subagent_routed(session, "claude", task) tui.exit_normally() task.assert_completed(session, "claude") + + +def test_smart_routing_claude_explicit_model_bypasses_routing(live_session, workspace): + """Scenario: request an explicit model while launching an interactive routing session. + + Expected: claude completes the real task with the caller's model choice, + starts no routing wrapper, exits normally, and reopens with usable input. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + model = session.model_for_explicit_case("claude") + command = [str(session.binary), "claude", "--enable-smart-routing", "--model", model] + with AgentTerminal(session, "claude", command, "explicit-model") as tui: + tui.boot() + tui.submit(task.prompt) + tui.wait_for_task(task) + tui.exit_normally() + task.assert_completed(session, "claude") + session.assert_not_routed() + with AgentTerminal(session, "claude", command, "reopen") as tui: + tui.boot() + tui.check_input_and_exit() + session.assert_not_routed() diff --git a/tests/integration/test_smart_routing_codex.py b/tests/integration/test_smart_routing_codex.py index c3bfd712c..64045d8e3 100644 --- a/tests/integration/test_smart_routing_codex.py +++ b/tests/integration/test_smart_routing_codex.py @@ -1,10 +1,10 @@ -"""Main CUJs: real first-prompt and child-task routing in Codex's TUI.""" +"""CUJs: real first-prompt and child-task routing in Codex's TUI.""" import pytest -from evidence import FileTask, assert_subagent_routed -from terminal import AgentTerminal +from utils.evidence import FileTask, assert_subagent_routed +from utils.terminal import AgentTerminal -pytestmark = [pytest.mark.live, pytest.mark.main, pytest.mark.tui, pytest.mark.codex] +pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.codex] def test_smart_routing_codex_first_prompt(live_session, workspace): @@ -37,6 +37,9 @@ def test_smart_routing_codex_first_prompt(live_session, workspace): assert "selection failed" not in log and "settings/update timed out" not in log, log tui.exit_normally() task.assert_completed(session, "codex") + with AgentTerminal(session, "codex", command, "reopen") as tui: + tui.boot() + tui.check_input_and_exit() def test_smart_routing_codex_subagent(live_session, workspace): @@ -44,7 +47,7 @@ def test_smart_routing_codex_subagent(live_session, workspace): Expected: a real child session has a correlated gateway routing decision, the child returns the file value, and the parent returns that result. - Child model identity remains explicitly unknown if its event omits it. + The native child turn must use the routed model; inherited parent turns do not count. """ session = live_session task = FileTask(session) @@ -65,6 +68,39 @@ def test_smart_routing_codex_subagent(live_session, workspace): tui.submit(task.delegate_prompt) tui.wait_for_task(task, timeout=240) task.assert_completed(session, "codex", child=True) - assert_subagent_routed(session, "codex") + assert_subagent_routed(session, "codex", task) tui.exit_normally() task.assert_completed(session, "codex") + + +def test_smart_routing_codex_explicit_model_bypasses_routing(live_session, workspace): + """Scenario: request an explicit model while launching an interactive routing session. + + Expected: codex completes the real task with the caller's model choice, + starts no routing wrapper, exits normally, and reopens with usable input. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + model = session.model_for_explicit_case("codex") + command = [str(session.binary), "codex", "--enable-smart-routing", "--", "--model", model] + with AgentTerminal(session, "codex", command, "explicit-model") as tui: + tui.boot() + tui.submit(task.prompt) + tui.wait_for_task(task) + tui.exit_normally() + task.assert_completed(session, "codex") + session.assert_not_routed() + with AgentTerminal(session, "codex", command, "reopen") as tui: + tui.boot() + tui.check_input_and_exit() + session.assert_not_routed() diff --git a/tests/integration/test_ug_claude_commands.py b/tests/integration/test_ug_claude_commands.py new file mode 100644 index 000000000..84cfbaf5f --- /dev/null +++ b/tests/integration/test_ug_claude_commands.py @@ -0,0 +1,55 @@ +"""CUJs for inspecting claude commands through ug.""" + +import pytest + +pytestmark = [pytest.mark.live, pytest.mark.claude] + + +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_claude_auth_help(live_session, workspace, routing): + """Scenario: configure claude, then ask ug for auth subcommand help. + + Expected: the real claude help is returned, with no routing wrapper or + model request. This verifies command dispatch, not an interactive session. + """ + session = live_session + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + expected = session.run("auth", "--help", binary="claude").stdout.strip() + actual = session.run("claude", "--", "auth", "--help").stdout + assert expected and expected in actual, actual + session.assert_not_routed() + + +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_claude_mcp_help(live_session, workspace, routing): + """Scenario: configure claude, then ask ug for mcp subcommand help. + + Expected: the real claude help is returned, with no routing wrapper or + model request. This verifies command dispatch, not an interactive session. + """ + session = live_session + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + expected = session.run("mcp", "--help", binary="claude").stdout.strip() + actual = session.run("claude", "--", "mcp", "--help").stdout + assert expected and expected in actual, actual + session.assert_not_routed() diff --git a/tests/integration/test_ug_claude_headless.py b/tests/integration/test_ug_claude_headless.py new file mode 100644 index 000000000..0f537838a --- /dev/null +++ b/tests/integration/test_ug_claude_headless.py @@ -0,0 +1,225 @@ +"""CUJs for using claude from scripts through installed ug.""" + +import json + +import pytest +from utils.evidence import FileTask + +pytestmark = [pytest.mark.live, pytest.mark.claude] + + +def test_ug_claude_headless_prompt_argument(live_session, workspace): + """Scenario: configure claude and submit a headless prompt via argument. + + Expected: the real agent reads the fixture and returns its unknown value in + its structured completed answer, with exit code zero. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + result = session.run( + "claude", + "--", + "-p", + task.prompt, + "--output-format", + "json", + "--allowedTools", + "Read", + timeout=180, + ) + task.assert_headless_answer("claude", result) + session.assert_not_routed() + + +def test_ug_claude_headless_prompt_stdin(live_session, workspace): + """Scenario: configure claude and submit a headless prompt via stdin. + + Expected: the real agent reads the fixture and returns its unknown value in + its structured completed answer, with exit code zero. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + result = session.run( + "claude", + "--", + "-p", + "--output-format", + "json", + "--allowedTools", + "Read", + timeout=180, + input_text=task.prompt + "\n", + ) + task.assert_headless_answer("claude", result) + session.assert_not_routed() + + +def test_ug_claude_headless_prompt_after_separator(live_session, workspace): + """Scenario: configure claude and submit a headless prompt via after separator. + + Expected: the real agent reads the fixture and returns its unknown value in + its structured completed answer, with exit code zero. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + result = session.run( + "claude", + "--", + "-p", + "--output-format", + "json", + "--allowedTools", + "Read", + "--", + task.prompt, + timeout=180, + ) + task.assert_headless_answer("claude", result) + session.assert_not_routed() + + +@pytest.mark.parametrize("model_form", ["separate", "equals"]) +def test_ug_claude_headless_explicit_model_bypasses_routing(live_session, workspace, model_form): + """Scenario: choose an explicit model while global smart routing is enabled. + + Expected: the model option is accepted, the real file task completes, and + no routing wrapper overrides the caller's choice. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + model = session.model_for_explicit_case("claude") + model_args = ["--model", model] if model_form == "separate" else [f"--model={model}"] + session.env["ENABLE_SMART_ROUTING_V2"] = "1" + result = session.run( + "claude", + "--", + "-p", + task.prompt, + "--output-format", + "json", + "--allowedTools", + "Read", + *model_args, + timeout=180, + ) + task.assert_headless_answer("claude", result) + session.assert_not_routed() + + +def test_ug_claude_preserves_caller_settings_and_hook(live_session, workspace): + """Scenario: a launcher passes a settings path containing spaces to ug claude. + + Expected: the caller's real SessionStart hook executes, its input file stays + unchanged, and gateway authentication still supports a completed file task. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + settings = session.cwd / "caller settings.json" + # Ordinary user-owned input, not fabricated ug state or generated gateway config. + content = json.dumps( + { + "hooks": { + "SessionStart": [ + { + "hooks": [ + {"type": "command", "command": "echo caller-hook-ran > caller-hook.txt"} + ] + } + ] + } + } + ) + settings.write_text(content) + result = session.run( + "claude", + "--", + "-p", + task.prompt, + "--output-format", + "json", + "--allowedTools", + "Read", + "--settings", + str(settings), + timeout=180, + ) + task.assert_headless_answer("claude", result) + assert (session.cwd / "caller-hook.txt").read_text().strip() == "caller-hook-ran" + assert settings.read_text() == content + + +def test_ug_claude_reports_unsupported_short_model_option(live_session, workspace): + """Scenario: pass -m to the selected Claude version, which does not support it. + + Expected: ug preserves the actual agent's unknown-option error and status. + This is an error-reporting journey, not a successful inference claim. + """ + session = live_session + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + expected = session.run("-m", "sonnet", "-p", "hi", binary="claude", ok=False) + actual = session.run("claude", "--", "-m", "sonnet", "-p", "hi", ok=False) + assert expected.returncode != 0 and "unknown option '-m'" in expected.stderr + assert actual.returncode == expected.returncode + assert "unknown option '-m'" in actual.stderr diff --git a/tests/integration/test_ug_codex_app_server.py b/tests/integration/test_ug_codex_app_server.py new file mode 100644 index 000000000..9a9c3559b --- /dev/null +++ b/tests/integration/test_ug_codex_app_server.py @@ -0,0 +1,34 @@ +"""CUJ: a client connects to Codex's real app-server through ug.""" + +import pytest + +pytestmark = [pytest.mark.live, pytest.mark.codex] + + +@pytest.mark.parametrize("separator", [False, True], ids=["direct", "launcher-separator"]) +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_app_server_client_initializes(live_session, workspace, separator, routing): + """Scenario: configure Codex and connect a real stdio client to ug codex app-server. + + Expected: initialize returns a valid JSON-RPC result, diagnostics stay off + the protocol stream, and this utility command never starts smart routing. + """ + session = live_session + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + + args = ["app-server", "--listen", "stdio://"] + if separator: + args.insert(0, "--") + response = session.app_server_handshake(args) + assert response["result"]["userAgent"] + session.assert_not_routed() diff --git a/tests/integration/test_ug_codex_commands.py b/tests/integration/test_ug_codex_commands.py new file mode 100644 index 000000000..2e6a32de7 --- /dev/null +++ b/tests/integration/test_ug_codex_commands.py @@ -0,0 +1,134 @@ +"""CUJs for inspecting codex commands through ug.""" + +import pytest + +pytestmark = [pytest.mark.live, pytest.mark.codex] + + +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_app_help(live_session, workspace, routing): + """Scenario: configure codex, then ask ug for app subcommand help. + + Expected: the real codex help is returned, with no routing wrapper or + model request. This verifies command dispatch, not an interactive session. + """ + session = live_session + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + expected = session.run("app", "--help", binary="codex").stdout.strip() + actual = session.run("codex", "--", "app", "--help").stdout + assert expected and expected in actual, actual + session.assert_not_routed() + + +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_app_server_help(live_session, workspace, routing): + """Scenario: configure codex, then ask ug for app-server subcommand help. + + Expected: the real codex help is returned, with no routing wrapper or + model request. This verifies command dispatch, not an interactive session. + """ + session = live_session + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + expected = session.run("app-server", "--help", binary="codex").stdout.strip() + actual = session.run("codex", "--", "app-server", "--help").stdout + assert expected and expected in actual, actual + session.assert_not_routed() + + +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_exec_help(live_session, workspace, routing): + """Scenario: configure codex, then ask ug for exec subcommand help. + + Expected: the real codex help is returned, with no routing wrapper or + model request. This verifies command dispatch, not an interactive session. + """ + session = live_session + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + expected = session.run("exec", "--help", binary="codex").stdout.strip() + actual = session.run("codex", "--", "exec", "--help").stdout + assert expected and expected in actual, actual + session.assert_not_routed() + + +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_mcp_help(live_session, workspace, routing): + """Scenario: configure codex, then ask ug for mcp subcommand help. + + Expected: the real codex help is returned, with no routing wrapper or + model request. This verifies command dispatch, not an interactive session. + """ + session = live_session + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + expected = session.run("mcp", "--help", binary="codex").stdout.strip() + actual = session.run("codex", "--", "mcp", "--help").stdout + assert expected and expected in actual, actual + session.assert_not_routed() + + +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_app_reports_unknown_argument(live_session, workspace, routing): + """Scenario: pass an unknown option directly to ug codex app. + + Expected: the actual Codex parser's error and exit status are preserved, + without opening a desktop application or starting routing. + """ + session = live_session + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + args = ["app", "--ug-integration-unknown-option"] + expected = session.run(*args, binary="codex", ok=False) + actual = session.run("codex", *args, ok=False) + assert expected.returncode != 0 and "error:" in expected.stderr + assert actual.returncode == expected.returncode + parser_error = expected.stderr[expected.stderr.index("error:") :].strip() + assert parser_error in actual.stderr, actual.stdout + actual.stderr + session.assert_not_routed() diff --git a/tests/integration/test_ug_codex_headless.py b/tests/integration/test_ug_codex_headless.py new file mode 100644 index 000000000..0b83f54d1 --- /dev/null +++ b/tests/integration/test_ug_codex_headless.py @@ -0,0 +1,129 @@ +"""CUJs for using codex from scripts through installed ug.""" + +import pytest +from utils.evidence import FileTask + +pytestmark = [pytest.mark.live, pytest.mark.codex] + + +def test_ug_codex_headless_prompt_argument(live_session, workspace): + """Scenario: configure codex and submit a headless prompt via argument. + + Expected: the real agent reads the fixture and returns its unknown value in + its structured completed answer, with exit code zero. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + result = session.run( + "codex", "--", "exec", "--skip-git-repo-check", "--json", task.prompt, timeout=180 + ) + task.assert_headless_answer("codex", result) + session.assert_not_routed() + + +def test_ug_codex_headless_prompt_stdin(live_session, workspace): + """Scenario: configure codex and submit a headless prompt via stdin. + + Expected: the real agent reads the fixture and returns its unknown value in + its structured completed answer, with exit code zero. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + result = session.run( + "codex", + "--", + "exec", + "--skip-git-repo-check", + "--json", + "-", + timeout=180, + input_text=task.prompt + "\n", + ) + task.assert_headless_answer("codex", result) + session.assert_not_routed() + + +def test_ug_codex_headless_prompt_after_separator(live_session, workspace): + """Scenario: configure codex and submit a headless prompt via after separator. + + Expected: the real agent reads the fixture and returns its unknown value in + its structured completed answer, with exit code zero. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + result = session.run( + "codex", "--", "exec", "--skip-git-repo-check", "--json", "--", task.prompt, timeout=180 + ) + task.assert_headless_answer("codex", result) + session.assert_not_routed() + + +@pytest.mark.parametrize("model_form", ["separate", "equals", "short"]) +def test_ug_codex_headless_explicit_model_bypasses_routing(live_session, workspace, model_form): + """Scenario: choose an explicit model while global smart routing is enabled. + + Expected: the model option is accepted, the real file task completes, and + no routing wrapper overrides the caller's choice. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + model = session.model_for_explicit_case("codex") + model_args = ["--model", model] if model_form == "separate" else [f"--model={model}"] + if model_form == "short": + model_args = ["-m", model] + session.env["ENABLE_SMART_ROUTING_V2"] = "1" + result = session.run( + "codex", + "--", + "exec", + "--skip-git-repo-check", + "--json", + task.prompt, + *model_args, + timeout=180, + ) + task.assert_headless_answer("codex", result) + session.assert_not_routed() diff --git a/tests/integration/test_ug_configure_claude.py b/tests/integration/test_ug_configure_claude.py index e57488463..7d0942733 100644 --- a/tests/integration/test_ug_configure_claude.py +++ b/tests/integration/test_ug_configure_claude.py @@ -1,10 +1,10 @@ -"""Main CUJs: configure Claude through ug, then use its real interactive session.""" +"""CUJs: configure Claude through ug, then use its real interactive session.""" import pytest -from evidence import FileTask -from terminal import AgentTerminal, ConfigureTerminal +from utils.evidence import FileTask +from utils.terminal import AgentTerminal, ConfigureTerminal -pytestmark = [pytest.mark.live, pytest.mark.main, pytest.mark.tui, pytest.mark.claude] +pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.claude] def test_ug_configure_claude_databricks(live_session, workspace): diff --git a/tests/integration/test_ug_configure_claude_lifecycle.py b/tests/integration/test_ug_configure_claude_lifecycle.py new file mode 100644 index 000000000..666e95a59 --- /dev/null +++ b/tests/integration/test_ug_configure_claude_lifecycle.py @@ -0,0 +1,106 @@ +"""CUJs for changing and undoing claude's ug setup.""" + +import json + +import pytest +from utils.evidence import FileTask +from utils.terminal import TerminalProcess + +pytestmark = [pytest.mark.live, pytest.mark.claude] + + +def test_ug_configure_claude_repeat_and_revert(live_session, workspace): + """Scenario: configure twice over user-owned settings, use the agent, then revert. + + Expected: repeat setup remains usable, unrelated settings survive, no bearer + is saved in ug state, and repeated revert restores the original configuration. + The generated ug file must be removed; a leftover file remains a failure. + """ + session = live_session + task = FileTask(session) + # Ordinary pre-existing user settings, not manufactured gateway state. + user_path = session.home / ".claude/settings.json" + original = '{"permissions":{"allow":["Read"]},"env":{"UG_USER_SETTING":"keep"}}\n' + user_path.parent.mkdir(parents=True) + user_path.write_text(original) + + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + first = session.workspace_state() + assert "claude" in first["available_tools"] + assert first["claude_models"] + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + assert session.workspace_state()["available_tools"] == first["available_tools"] + assert workspace in session.run("status").stdout + current = json.loads(user_path.read_text()) + assert all(current.get(key) == value for key, value in json.loads(original).items()) + bearer_was_saved = any( + session.env["DATABRICKS_BEARER"] in path.read_text(errors="replace") + for path in (session.home / ".ucode").rglob("*") + if path.is_file() + ) + assert not bearer_was_saved, "The workspace bearer was saved in ug state" + + result = session.run( + "claude", + "--", + "-p", + task.prompt, + "--output-format", + "json", + "--allowedTools", + "Read", + timeout=180, + ) + task.assert_headless_answer("claude", result) + with TerminalProcess(session, "ug", [str(session.binary), "revert"], "revert") as terminal: + terminal.finish() + assert json.loads(user_path.read_text()) == json.loads(original) + assert "Not Configured" in session.run("status").stdout + assert not (session.home / ".claude/ucode-settings.json").exists() + with TerminalProcess( + session, "ug", [str(session.binary), "revert"], "repeat-revert" + ) as terminal: + terminal.finish() + + +def test_ug_configure_claude_rejects_invalid_credentials(live_session, workspace): + """Scenario: configure claude against the real workspace with a rejected bearer. + + Expected: authentication fails clearly and no successful setup is saved. + """ + session = live_session + session.env["DATABRICKS_BEARER"] = "ug-integration-intentionally-invalid" + result = session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ok=False, + ) + assert result.returncode != 0 + output = result.stdout + result.stderr + assert "rejected the access token" in output or "401" in output, output + assert "Configuration Complete" not in output + assert not (session.home / ".ucode/state.json").exists() diff --git a/tests/integration/test_ug_configure_codex.py b/tests/integration/test_ug_configure_codex.py index fe0513d0c..07ffa9556 100644 --- a/tests/integration/test_ug_configure_codex.py +++ b/tests/integration/test_ug_configure_codex.py @@ -1,10 +1,10 @@ -"""Main CUJs: configure Codex through ug, then use its real interactive session.""" +"""CUJs: configure Codex through ug, then use its real interactive session.""" import pytest -from evidence import FileTask -from terminal import AgentTerminal, ConfigureTerminal +from utils.evidence import FileTask +from utils.terminal import AgentTerminal, ConfigureTerminal -pytestmark = [pytest.mark.live, pytest.mark.main, pytest.mark.tui, pytest.mark.codex] +pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.codex] def test_ug_configure_codex_databricks(live_session, workspace): diff --git a/tests/integration/test_ug_configure_codex_lifecycle.py b/tests/integration/test_ug_configure_codex_lifecycle.py new file mode 100644 index 000000000..49bb3351b --- /dev/null +++ b/tests/integration/test_ug_configure_codex_lifecycle.py @@ -0,0 +1,98 @@ +"""CUJs for changing and undoing codex's ug setup.""" + +import tomllib + +import pytest +from utils.evidence import FileTask +from utils.terminal import TerminalProcess + +pytestmark = [pytest.mark.live, pytest.mark.codex] + + +def test_ug_configure_codex_repeat_and_revert(live_session, workspace): + """Scenario: configure twice over user-owned settings, use the agent, then revert. + + Expected: repeat setup remains usable, unrelated settings survive, no bearer + is saved in ug state, and repeated revert restores the original configuration. + The generated ug file must be removed; a leftover file remains a failure. + """ + session = live_session + task = FileTask(session) + # Ordinary pre-existing user settings, not manufactured gateway state. + user_path = session.home / ".codex/config.toml" + original = "# user comment\n[notice]\nhide_rate_limit_model_nudge = true\n" + user_path.parent.mkdir(parents=True) + user_path.write_text(original) + + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + first = session.workspace_state() + assert "codex" in first["available_tools"] + assert first["codex_models"] + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + assert session.workspace_state()["available_tools"] == first["available_tools"] + assert workspace in session.run("status").stdout + current = tomllib.loads(user_path.read_text()) + assert all(current.get(key) == value for key, value in tomllib.loads(original).items()) + bearer_was_saved = any( + session.env["DATABRICKS_BEARER"] in path.read_text(errors="replace") + for path in (session.home / ".ucode").rglob("*") + if path.is_file() + ) + assert not bearer_was_saved, "The workspace bearer was saved in ug state" + + result = session.run( + "codex", "--", "exec", "--skip-git-repo-check", "--json", task.prompt, timeout=180 + ) + task.assert_headless_answer("codex", result) + with TerminalProcess(session, "ug", [str(session.binary), "revert"], "revert") as terminal: + terminal.finish() + assert tomllib.loads(user_path.read_text()) == tomllib.loads(original) + assert "Not Configured" in session.run("status").stdout + assert not (session.home / ".codex/ucode.config.toml").exists() + with TerminalProcess( + session, "ug", [str(session.binary), "revert"], "repeat-revert" + ) as terminal: + terminal.finish() + + +def test_ug_configure_codex_rejects_invalid_credentials(live_session, workspace): + """Scenario: configure codex against the real workspace with a rejected bearer. + + Expected: authentication fails clearly and no successful setup is saved. + """ + session = live_session + session.env["DATABRICKS_BEARER"] = "ug-integration-intentionally-invalid" + result = session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ok=False, + ) + assert result.returncode != 0 + output = result.stdout + result.stderr + assert "rejected the access token" in output or "401" in output, output + assert "Configuration Complete" not in output + assert not (session.home / ".ucode/state.json").exists() diff --git a/tests/integration/Dockerfile b/tests/integration/utils/Dockerfile similarity index 93% rename from tests/integration/Dockerfile rename to tests/integration/utils/Dockerfile index 42efa0ed1..537818223 100644 --- a/tests/integration/Dockerfile +++ b/tests/integration/utils/Dockerfile @@ -1,4 +1,4 @@ -# Build from the repository root. Only the runner and its tests enter the image; +# Build using the documented minimal archive. Only the runner and tests enter the image; # pass a release version or mount a prebuilt wheel at runtime. FROM ghcr.io/astral-sh/uv:0.9.8 AS uv FROM node:22.19.0-bookworm-slim AS node diff --git a/tests/integration/Dockerfile.dockerignore b/tests/integration/utils/Dockerfile.dockerignore similarity index 100% rename from tests/integration/Dockerfile.dockerignore rename to tests/integration/utils/Dockerfile.dockerignore diff --git a/tests/integration/utils/__init__.py b/tests/integration/utils/__init__.py new file mode 100644 index 000000000..d452b37b6 --- /dev/null +++ b/tests/integration/utils/__init__.py @@ -0,0 +1 @@ +"""Process, terminal, and evidence utilities for the integration CUJs.""" diff --git a/tests/integration/evidence.py b/tests/integration/utils/evidence.py similarity index 50% rename from tests/integration/evidence.py rename to tests/integration/utils/evidence.py index a6684e1b3..6cfafdac7 100644 --- a/tests/integration/evidence.py +++ b/tests/integration/utils/evidence.py @@ -90,8 +90,32 @@ def assert_completed(self, session, agent: str, *, child: bool = False) -> None: "echoed prompts and tool results do not count as completed answers." ) - -def assert_subagent_routed(session, agent: str) -> None: + def assert_headless_answer(self, agent: str, result) -> None: + """Read the real CLI's structured final answer, never its echoed input.""" + payloads = [] + for line in result.stdout.splitlines(): + try: + value = json.loads(line) + except json.JSONDecodeError: + continue # ug may print human-readable launch status before agent JSON. + if isinstance(value, dict): + payloads.append(value) + if agent == "claude": + final = [row for row in payloads if row.get("type") == "result"] + assert final and not final[-1].get("is_error"), result.stdout + assert self.value in final[-1].get("result", ""), result.stdout + else: + assert any(row.get("type") == "turn.completed" for row in payloads), result.stdout + answers = [ + row.get("item", {}).get("text", "") + for row in payloads + if row.get("type") == "item.completed" + and row.get("item", {}).get("type") == "agent_message" + ] + assert any(self.value in answer for answer in answers), result.stdout + + +def assert_subagent_routed(session, agent: str, task: FileTask) -> None: """Require a real gateway decision correlated with an actual child start.""" root = session.home / ".ucode" decisions = read_jsonl(root / f"{agent}-smart-routing-decisions.jsonl") @@ -100,6 +124,70 @@ def assert_subagent_routed(session, agent: str) -> None: assert decisions, "No real subagent routing decision was recorded" for decision in decisions: assert decision.get("requested_model") and decision.get("router_model"), decision + if agent == "codex": + # Codex exposes parent linkage and the child's actual turn model in its + # native rollouts. Its ug SubagentStart audit can be empty even when the + # child ran. Match native evidence, including the completed file task. + sessions = agent_sessions(session, agent) + linked = [] + for path, records in sessions.items(): + metadata = next( + (row["payload"] for row in records if row.get("type") == "session_meta"), {} + ) + source = metadata.get("source") + if not isinstance(source, dict): + continue + parent_id = source.get("subagent", {}).get("thread_spawn", {}).get("parent_thread_id") + if not parent_id: + continue + # A child rollout starts with inherited parent history. Exclude + # those turn IDs so a parent's answer/model cannot satisfy this check. + parent_turn_ids = set() + for other in sessions.values(): + first_meta = next( + (row["payload"] for row in other if row.get("type") == "session_meta"), {} + ) + if first_meta.get("id") == parent_id: + parent_turn_ids.update( + row["payload"]["turn_id"] + for row in other + if row.get("type") == "turn_context" and row["payload"].get("turn_id") + ) + assert parent_turn_ids, f"No native parent turns found for {parent_id}" + for decision in decisions: + if decision.get("session_id") != parent_id: + continue + routed_turn_ids = { + row["payload"]["turn_id"] + for row in records + if row.get("type") == "turn_context" + and row["payload"].get("model") == decision["requested_model"] + and row["payload"].get("turn_id") + } - parent_turn_ids + for row in records: + payload = row.get("payload", {}) + if ( + row.get("type") == "event_msg" + and payload.get("type") == "task_complete" + and payload.get("turn_id") in routed_turn_ids + and task.value in (payload.get("last_agent_message") or "") + ): + linked.append( + { + "decision_id": decision["decision_id"], + "parent_id": parent_id, + "child_id": metadata["id"], + "path": path, + "turn_id": payload["turn_id"], + "model": decision["requested_model"], + } + ) + session.record("subagent-routing.json", {"decisions": decisions, "native_children": linked}) + assert linked, "No routed native child turn completed the delegated file task" + assert {row["decision_id"] for row in linked} == { + row["decision_id"] for row in decisions + }, "A routing decision had no matching completed child turn" + return decision_ids = {decision["decision_id"] for decision in decisions} routed_starts = [row for row in audit if row.get("decision_id") in decision_ids] assert routed_starts and all(row.get("agent_id") for row in routed_starts), audit diff --git a/tests/integration/harness.py b/tests/integration/utils/harness.py similarity index 96% rename from tests/integration/harness.py rename to tests/integration/utils/harness.py index c85f60f73..cbf9c3590 100644 --- a/tests/integration/harness.py +++ b/tests/integration/utils/harness.py @@ -143,19 +143,6 @@ def run( def record(self, name: str, value: object) -> None: (self.artifacts / name).write_text(self.redact(json.dumps(value, indent=2))) - def configure(self, agent: str, workspace: str, *, ok: bool = True): - return self.run( - "configure", - "--agents", - agent, - "--workspaces", - workspace, - "--skip-validate", - "--skip-upgrade", - "--disable-databricks-ai-tools", - ok=ok, - ) - def state(self) -> dict: return json.loads((self.home / ".ucode/state.json").read_text()) diff --git a/tests/integration/terminal.py b/tests/integration/utils/terminal.py similarity index 99% rename from tests/integration/terminal.py rename to tests/integration/utils/terminal.py index f29bf178a..ea6b05208 100644 --- a/tests/integration/terminal.py +++ b/tests/integration/utils/terminal.py @@ -15,7 +15,8 @@ import pexpect import pyte -from evidence import agent_sessions + +from .evidence import agent_sessions class TerminalScreen(pyte.Screen): From eb82a05a209cf01a73e82fa899bef4166e9ad28b Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Mon, 14 Sep 2026 11:36:49 -0400 Subject: [PATCH 5/6] Keep app-server stdout clean and delete generated configs on revert Two product bugs surfaced by the integration suite: - `ug codex app-server --listen stdio://` printed status lines like "Databricks auth already available" to stdout, corrupting the JSON-RPC stream. ug's Rich output now moves to stderr for that launch; fd 1 stays untouched for the exec'd agent. - A second configure or a launch-time model-preference clear backed up the config file ucode itself generated, so `ug revert` restored that snapshot instead of deleting it. Backups now only capture files that predate ucode's management of the tool, so revert removes `~/.claude/ucode-settings.json` and `~/.codex/ucode.config.toml`. Co-authored-by: Isaac --- src/ucode/agents/claude.py | 14 ++++++++++-- src/ucode/agents/codex.py | 13 ++++++++--- src/ucode/cli.py | 15 +++++++++++++ src/ucode/state.py | 10 +++++++++ src/ucode/ui.py | 13 +++++++++++ tests/test_agent_claude.py | 37 ++++++++++++++++++++++++++++++ tests/test_agent_codex.py | 46 ++++++++++++++++++++++++++++++++++++++ tests/test_cli.py | 25 +++++++++++++++++++++ 8 files changed, 168 insertions(+), 5 deletions(-) diff --git a/src/ucode/agents/claude.py b/src/ucode/agents/claude.py index a77013d10..e54240be9 100644 --- a/src/ucode/agents/claude.py +++ b/src/ucode/agents/claude.py @@ -49,7 +49,13 @@ remove_smart_routing_hooks, sync_smart_routing_hooks, ) -from ucode.state import MANAGED_OVERLAY_KEY, get_provider_service, mark_tool_managed, save_state +from ucode.state import ( + MANAGED_OVERLAY_KEY, + get_provider_service, + is_tool_managed, + mark_tool_managed, + save_state, +) from ucode.telemetry import agent_version, ucode_version from ucode.tracing import tracing_env from ucode.ui import print_note, print_success, print_warning @@ -575,7 +581,11 @@ def write_tool_config( custom_model: str | None = None, coding_agent_config_defaults: dict[str, str] | None = None, ) -> dict: - backup_existing_file(CLAUDE_SETTINGS_PATH, CLAUDE_BACKUP_PATH) + # Back up only a file that predates ucode's management of the tool. A + # re-configure would otherwise snapshot ucode's own generated file, and + # revert would restore that snapshot instead of deleting the file. + if not is_tool_managed(state, "claude"): + backup_existing_file(CLAUDE_SETTINGS_PATH, CLAUDE_BACKUP_PATH) web_search_model = _resolve_web_search_model(state) # Relayed inference points at a local refresh proxy; its loopback base URL is # recorded in state so launch starts the proxy on the matching port. diff --git a/src/ucode/agents/codex.py b/src/ucode/agents/codex.py index f0f104292..67b89b6e9 100644 --- a/src/ucode/agents/codex.py +++ b/src/ucode/agents/codex.py @@ -46,7 +46,7 @@ sync_smart_routing_hooks, ) from ucode.smart_routing.codex_routing import codex_model_id -from ucode.state import mark_tool_managed, save_state +from ucode.state import is_tool_managed, mark_tool_managed, save_state from ucode.telemetry import agent_version, ucode_version from ucode.ui import print_warning_err @@ -351,7 +351,11 @@ def write_tool_config(state: dict, model: str | None = None, provider: str | Non return state _remove_legacy_ucode_profile() - backup_existing_file(CODEX_CONFIG_PATH, CODEX_BACKUP_PATH) + # Back up only a file that predates ucode's management of the tool. A + # re-configure would otherwise snapshot ucode's own generated file, and + # revert would restore that snapshot instead of deleting the file. + if not is_tool_managed(state, "codex"): + backup_existing_file(CODEX_CONFIG_PATH, CODEX_BACKUP_PATH) overlay = render_overlay( workspace, chosen_model, @@ -492,7 +496,10 @@ def clear_model_preferences(state: dict) -> bool: doc.pop(key) changed = True if changed: - backup_existing_file(CODEX_CONFIG_PATH, CODEX_BACKUP_PATH) + # Never snapshot ucode's own generated file here; revert would restore + # the snapshot instead of deleting the file. + if not is_tool_managed(state, "codex"): + backup_existing_file(CODEX_CONFIG_PATH, CODEX_BACKUP_PATH) write_toml_file(CODEX_CONFIG_PATH, doc) return changed diff --git a/src/ucode/cli.py b/src/ucode/cli.py index c52944079..4ffb40625 100644 --- a/src/ucode/cli.py +++ b/src/ucode/cli.py @@ -149,6 +149,7 @@ prompt_for_tools, prompt_for_workspace, prompt_yes_no, + redirect_output_to_stderr, set_verbosity, spinner, status_badge, @@ -1981,6 +1982,16 @@ def _download_managed_skills(managed: dict, state: dict) -> None: print_note(f"Downloaded workspace skill(s) to disk: {', '.join(written)}") +def _child_owns_stdout(tool: str, tool_args: list[str]) -> bool: + """True when the forwarded agent command speaks a stdio protocol on stdout. + + ``codex app-server`` puts its JSON-RPC stream on stdout, so ug's status + output must move to stderr for that launch; the file descriptor stays + untouched for the agent process. + """ + return tool == "codex" and tool_args[:1] == ["app-server"] + + def _should_launch_smart_routing( tool: str, tool_args: list[str], @@ -2038,6 +2049,10 @@ def _launch_tool( ) -> None: try: tool = normalize_tool(tool_name) + # Before any status print: a stdio-protocol subcommand owns stdout, so + # every ug line from here on must go to stderr instead. + if _child_owns_stdout(tool, ctx.args): + redirect_output_to_stderr() explicit_prompt = _has_explicit_prompt(ctx) smart_routing_enabled = smart_routing_v2.enabled() # Launchers such as isaac put their harness arguments after `--`, so the harness's own diff --git a/src/ucode/state.py b/src/ucode/state.py index c02855537..e37f60e3e 100644 --- a/src/ucode/state.py +++ b/src/ucode/state.py @@ -251,6 +251,16 @@ def clear_state() -> None: raise RuntimeError(f"Failed to clear state file: {STATE_PATH}") from exc +def is_tool_managed(state: dict, tool: str) -> bool: + """True once configure has written ucode's config for ``tool``. + + Used to distinguish a config file the user owned before ucode ran from one + ucode itself generated: only the former is backup-worthy, because revert + restores a backup in place instead of deleting ucode's generated file. + """ + return bool((state.get("managed_configs") or {}).get(tool)) + + def mark_tool_managed(state: dict, tool: str, managed_keys: list) -> dict: """Record which config keys ucode manages for ``tool``.""" managed_configs = dict(state.get("managed_configs") or {}) diff --git a/src/ucode/ui.py b/src/ucode/ui.py index ce92bcbcf..c2c4af57c 100644 --- a/src/ucode/ui.py +++ b/src/ucode/ui.py @@ -24,6 +24,19 @@ console = Console(highlight=False) err_console = Console(stderr=True, highlight=False) + +def redirect_output_to_stderr() -> None: + """Move status output to stderr because a child process now owns stdout. + + ``codex app-server`` speaks its JSON-RPC protocol on stdout, so any ug line + printed there corrupts the stream. Rich resolves ``console.file`` to + ``sys.stdout`` at print time, so rebinding the module attribute sends every + status print to stderr while ``os.execvp`` still hands the launched agent + the untouched stdout file descriptor. + """ + sys.stdout = sys.stderr + + # Past this many options the choice list is pinned to a fixed-height scrolling viewport (see # `_cap_choice_viewport`) rather than growing to fill the terminal, and the pickers append a # "↑/↓ scroll" note to their instruction line. The value is both the boundary and the number of diff --git a/tests/test_agent_claude.py b/tests/test_agent_claude.py index 92f99fe22..0444170fd 100644 --- a/tests/test_agent_claude.py +++ b/tests/test_agent_claude.py @@ -1585,3 +1585,40 @@ def test_missing_login_runs_browser_flow(self, monkeypatch): monkeypatch.setattr(claude, "print_success", lambda *a, **kw: None) claude._ensure_subscription_login() assert calls == [[claude.SPEC["binary"], "auth", "login"]] + + +class TestWriteToolConfigBackup: + """A re-configure must not snapshot the file ucode itself generated.""" + + def _patch(self, monkeypatch, tmp_path): + monkeypatch.setattr(claude, "CLAUDE_SETTINGS_PATH", tmp_path / "ucode-settings.json") + monkeypatch.setattr(claude, "CLAUDE_BACKUP_PATH", tmp_path / "backup.json") + monkeypatch.setattr("ucode.config_io.APP_DIR", tmp_path) + monkeypatch.setattr(claude, "save_state", lambda state: None) + monkeypatch.setattr(claude, "_register_web_search_mcp", lambda *a, **kw: None) + + def test_first_configure_backs_up_user_owned_file(self, tmp_path, monkeypatch): + self._patch(monkeypatch, tmp_path) + (tmp_path / "ucode-settings.json").write_text( + '{"permissions": {"allow": ["Read"]}}', encoding="utf-8" + ) + + claude.write_tool_config( + {"workspace": WS, "claude_models": {}}, "databricks-claude-sonnet-4" + ) + + backup = (tmp_path / "backup.json").read_text(encoding="utf-8") + assert backup == '{"permissions": {"allow": ["Read"]}}' + + def test_reconfigure_does_not_back_up_generated_file(self, tmp_path, monkeypatch): + self._patch(monkeypatch, tmp_path) + state = { + "workspace": WS, + "claude_models": {}, + # load_state after a first configure: ucode already manages this file. + "managed_configs": {"claude": {"keys": [["env", "ANTHROPIC_BASE_URL"]]}}, + } + + claude.write_tool_config(state, "databricks-claude-sonnet-4") + + assert not (tmp_path / "backup.json").exists() diff --git a/tests/test_agent_codex.py b/tests/test_agent_codex.py index 4f4af9709..ace56212f 100644 --- a/tests/test_agent_codex.py +++ b/tests/test_agent_codex.py @@ -732,3 +732,49 @@ def test_invalid_managed_toml_is_not_modified(self, tmp_path, monkeypatch): codex.write_tool_config({"workspace": WS, "codex_models": ["gpt-5"]}) assert managed_path.read_text(encoding="utf-8") == "[invalid" + + +class TestWriteConfigBackup: + """A re-configure must not snapshot the file ucode itself generated.""" + + def _patch(self, monkeypatch, tmp_path): + monkeypatch.setattr(codex, "CODEX_CONFIG_PATH", tmp_path / "ucode.config.toml") + monkeypatch.setattr(codex, "CODEX_BACKUP_PATH", tmp_path / "backup.toml") + monkeypatch.setattr("ucode.config_io.APP_DIR", tmp_path) + monkeypatch.setattr(codex, "agent_version", lambda binary: "0.134.0") + monkeypatch.setattr(codex, "save_state", lambda state: None) + + def test_first_configure_backs_up_user_owned_profile(self, tmp_path, monkeypatch): + self._patch(monkeypatch, tmp_path) + (tmp_path / "ucode.config.toml").write_text("# user comment\n", encoding="utf-8") + + codex.write_tool_config({"workspace": WS, "codex_models": ["gpt-5"]}) + + assert (tmp_path / "backup.toml").read_text(encoding="utf-8") == "# user comment\n" + + def test_reconfigure_does_not_back_up_generated_profile(self, tmp_path, monkeypatch): + self._patch(monkeypatch, tmp_path) + state = { + "workspace": WS, + "codex_models": ["gpt-5"], + # load_state after a first configure: ucode already manages this file. + "managed_configs": {"codex": {"keys": [["model_provider"]]}}, + } + + codex.write_tool_config(state) + + assert not (tmp_path / "backup.toml").exists() + + def test_clear_model_preferences_does_not_back_up_generated_profile( + self, tmp_path, monkeypatch + ): + self._patch(monkeypatch, tmp_path) + (tmp_path / "ucode.config.toml").write_text('model = "system.ai.gpt-5"\n', encoding="utf-8") + + changed = codex.clear_model_preferences( + {"workspace": WS, "managed_configs": {"codex": {"keys": []}}} + ) + + assert changed is True + assert "model" not in read_toml_safe(tmp_path / "ucode.config.toml") + assert not (tmp_path / "backup.toml").exists() diff --git a/tests/test_cli.py b/tests/test_cli.py index 32e81f711..79f1f5a3a 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -3955,3 +3955,28 @@ def test_still_logs_in_when_nothing_external_is_set(self, monkeypatch): monkeypatch.delenv("DATABRICKS_BEARER_COMMAND", raising=False) assert self._run(monkeypatch) == ["https://ws.cloud.databricks.com"] + + +class TestStdioProtocolLaunch: + """`codex app-server` owns stdout, so ug's status output moves to stderr.""" + + def test_app_server_subcommand_owns_stdout(self): + assert cli_mod._child_owns_stdout("codex", ["app-server", "--listen", "stdio://"]) is True + + def test_other_codex_launches_keep_stdout(self): + assert cli_mod._child_owns_stdout("codex", []) is False + assert cli_mod._child_owns_stdout("codex", ["exec", "--json", "hi"]) is False + + def test_other_agents_never_own_stdout(self): + assert cli_mod._child_owns_stdout("claude", ["app-server"]) is False + assert cli_mod._child_owns_stdout("gemini", []) is False + + def test_redirect_rebinds_stdout_without_touching_the_descriptor(self): + import sys + + real_stdout = sys.stdout + try: + cli_mod.redirect_output_to_stderr() + assert sys.stdout is sys.stderr + finally: + sys.stdout = real_stdout From 8e94bc72f765ca1f5e46c9b1efb6bcf618491adb Mon Sep 17 00:00:00 2001 From: Rohit Agrawal Date: Mon, 14 Sep 2026 11:36:51 -0400 Subject: [PATCH 6/6] Raise MPS task and CI live-job timeouts The configure MPS CUJs can exceed a 180s task wait on a slow gateway round trip, and the live CI job's 45-minute timeout fired before the runner's own 3600s pytest deadline could produce its report. Co-authored-by: Isaac --- .github/workflows/integration.yml | 4 +++- tests/integration/test_ug_configure_claude.py | 2 +- tests/integration/test_ug_configure_codex.py | 2 +- 3 files changed, 5 insertions(+), 3 deletions(-) diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index c381a0b43..c097f70e9 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -115,7 +115,9 @@ jobs: name: CUJs needs: workspace runs-on: ubuntu-22.04 - timeout-minutes: 45 + # The runner enforces its own 3600s pytest deadline; keep the job above it + # so a slow run fails through the runner's report instead of a job kill. + timeout-minutes: 70 env: UCODE_TEST_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} diff --git a/tests/integration/test_ug_configure_claude.py b/tests/integration/test_ug_configure_claude.py index 7d0942733..37fc5ac5c 100644 --- a/tests/integration/test_ug_configure_claude.py +++ b/tests/integration/test_ug_configure_claude.py @@ -76,7 +76,7 @@ def test_ug_configure_claude_anthropic_mps(live_session, workspace, claude_provi ) as tui: tui.boot() tui.submit(task.prompt) - tui.wait_for_task(task) + tui.wait_for_task(task, timeout=300) tui.exit_normally() task.assert_completed(session, "claude") session.assert_not_routed() diff --git a/tests/integration/test_ug_configure_codex.py b/tests/integration/test_ug_configure_codex.py index 07ffa9556..c2759c604 100644 --- a/tests/integration/test_ug_configure_codex.py +++ b/tests/integration/test_ug_configure_codex.py @@ -70,7 +70,7 @@ def test_ug_configure_codex_openai_mps(live_session, workspace, codex_provider): with AgentTerminal(session, "codex", [str(session.binary), "codex"], "provider-session") as tui: tui.boot() tui.submit(task.prompt) - tui.wait_for_task(task) + tui.wait_for_task(task, timeout=300) tui.exit_normally() task.assert_completed(session, "codex") session.assert_not_routed()