diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml new file mode 100644 index 000000000..c097f70e9 --- /dev/null +++ b/.github/workflows/integration.yml @@ -0,0 +1,161 @@ +name: Integration + +on: + pull_request: + paths: ['src/**', 'tests/**', 'scripts/run_integration.py', 'pyproject.toml', 'uv.lock', '.github/workflows/integration.yml'] + push: + branches: [main] + paths: ['src/**', 'tests/**', 'scripts/run_integration.py', 'pyproject.toml', 'uv.lock', '.github/workflows/integration.yml'] + workflow_dispatch: + inputs: + suite: + description: Which integration checks to run + type: choice + options: [live, tui, installation] + default: live + ug_version: + description: Exact ucode release, or checkout + default: checkout + required: true + entry_point: + description: Console command (older releases may only have ucode) + type: choice + options: [ug, ucode] + default: ug + claude_version: + description: Exact Claude Code version + default: 2.1.268 + required: true + codex_version: + description: Exact Codex version + default: 0.154.0 + required: true + dependency: + description: Optional exact dependency for reproduction (e.g. tomlkit==0.14.0) + default: '' + required: false + index_url: + description: Python package index containing the requested ug version + default: https://pypi.org/simple + required: true + +permissions: + contents: read + +concurrency: + group: integration-${{ github.event.pull_request.number || github.ref }}-${{ github.event_name }} + cancel-in-progress: true + +env: + EXPECTED_WORKSPACE: https://eng-ml-inference-team-us-east-1.cloud.databricks.com + UG_VERSION: ${{ inputs.ug_version || 'checkout' }} + ENTRY_POINT: ${{ inputs.entry_point || 'ug' }} + CLAUDE_VERSION: ${{ inputs.claude_version || '2.1.268' }} + CODEX_VERSION: ${{ inputs.codex_version || '0.154.0' }} + PACKAGE_INDEX: ${{ inputs.index_url || 'https://pypi.org/simple' }} + +jobs: + installation: + name: Installation + runs-on: ubuntu-22.04 + timeout-minutes: 20 + steps: + - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + with: + fetch-depth: 0 + - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 + with: + node-version: 22.19.0 + - uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0 + with: + version: 0.9.8 + - name: Test a fresh installed package without credentials + run: | + uv run --no-project --python 3.12 python scripts/run_integration.py \ + --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ + --claude-version "$CLAUDE_VERSION" --codex-version "$CODEX_VERSION" \ + --index-url "$PACKAGE_INDEX" --installation-only --output "$RUNNER_TEMP/ug-integration" + - name: Upload installation evidence + if: ${{ !cancelled() }} + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 + with: + name: integration-installation + include-hidden-files: true + path: | + ${{ runner.temp }}/ug-integration/*.json + ${{ runner.temp }}/ug-integration/*.txt + ${{ runner.temp }}/ug-integration/*.xml + ${{ runner.temp }}/ug-integration/*.log + ${{ runner.temp }}/ug-integration/artifacts/ + ${{ runner.temp }}/ug-integration/wheels/ + + workspace: + name: Workspace + if: ${{ inputs.suite != 'installation' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) }} + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - name: Verify the existing e2e workspace and credential are configured + env: + UCODE_TEST_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} + DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} + run: | + python3 - <<'PY' + import os + expected = os.environ["EXPECTED_WORKSPACE"].rstrip("/") + actual = os.environ["UCODE_TEST_WORKSPACE"].strip().rstrip("/") + if actual != expected: + raise SystemExit("UCODE_TEST_WORKSPACE does not match the expected CI workspace.") + if not os.environ["DATABRICKS_BEARER"].strip(): + raise SystemExit("The existing e2e DATABRICKS_BEARER secret is missing.") + print("CI workspace matches; the existing e2e credential is present.") + PY + + live: + name: CUJs + needs: workspace + runs-on: ubuntu-22.04 + # The runner enforces its own 3600s pytest deadline; keep the job above it + # so a slow run fails through the runner's report instead of a job kill. + timeout-minutes: 70 + env: + UCODE_TEST_WORKSPACE: ${{ secrets.UCODE_TEST_WORKSPACE }} + DATABRICKS_BEARER: ${{ secrets.DATABRICKS_BEARER }} + DEPENDENCY: ${{ inputs.dependency }} + TEST_MARKER: ${{ inputs.suite || 'live' }} + steps: + - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + with: + fetch-depth: 0 + - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 + with: + node-version: 22.19.0 + - uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0 + with: + version: 0.9.8 + - uses: databricks/setup-cli@bdb89f81c11a5bd647fd55b585b7c396ec68a25a # v1.0.0 + - name: Run the selected live cases against the e2e workspace + shell: bash + run: | + args=() + if [[ -n "$DEPENDENCY" ]]; then + args+=(--dependency "$DEPENDENCY") + fi + uv run --no-project --python 3.12 python scripts/run_integration.py \ + --python 3.12 --ug-version "$UG_VERSION" --entry-point "$ENTRY_POINT" \ + --claude-version "$CLAUDE_VERSION" --codex-version "$CODEX_VERSION" \ + --index-url "$PACKAGE_INDEX" --output "$RUNNER_TEMP/ug-integration" \ + "${args[@]}" -- -m "$TEST_MARKER" + - name: Upload live test evidence + if: ${{ !cancelled() }} + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 + with: + name: integration-cujs + include-hidden-files: true + path: | + ${{ runner.temp }}/ug-integration/*.json + ${{ runner.temp }}/ug-integration/*.txt + ${{ runner.temp }}/ug-integration/*.xml + ${{ runner.temp }}/ug-integration/*.log + ${{ runner.temp }}/ug-integration/artifacts/ + ${{ runner.temp }}/ug-integration/wheels/ diff --git a/.gitignore b/.gitignore index 007899c40..c616aacf8 100644 --- a/.gitignore +++ b/.gitignore @@ -6,3 +6,4 @@ dist/ .venv/ .DS_Store .isaac/ +.integration-runs/ diff --git a/README.md b/README.md index de31ee2b4..c5981ce75 100644 --- a/README.md +++ b/README.md @@ -401,7 +401,13 @@ uv sync uv run ruff check . # lint ``` -4. For end-to-end testing against a real workspace: +4. For **integration tests** of installed ug and agent versions against the same + real workspace, see [the integration suite](tests/integration/README.md). + It uses separate processes and fresh homes, with no application mocks or + monkeypatching. The runner accepts ug/Claude/Codex versions and dependency + constraints to reproduce user issues. + + The existing e2e tests remain available separately: ```bash UCODE_TEST_WORKSPACE= uv run pytest tests/test_e2e.py -v diff --git a/scripts/run_integration.py b/scripts/run_integration.py new file mode 100644 index 000000000..a59c6b3d7 --- /dev/null +++ b/scripts/run_integration.py @@ -0,0 +1,539 @@ +"""Install a reproducible ug/agent combination and run black-box integration tests. + +Uses fresh virtualenvs and an isolated npm prefix, never the checkout's uv.lock +or the developer's installed agents. Only the live workspace is shared with e2e. +""" + +from __future__ import annotations + +import argparse +import contextlib +import datetime as dt +import hashlib +import json +import os +import platform +import re +import shutil +import signal +import subprocess +import sys +import xml.etree.ElementTree as ET +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +AGENT_PACKAGES = {"claude": "@anthropic-ai/claude-code", "codex": "@openai/codex"} + + +@contextlib.contextmanager +def managed_process(command, *, interrupt=False, **kwargs): + """Bound child lifetimes, including descendants that outlive their parent.""" + proc = subprocess.Popen(command, start_new_session=True, **kwargs) + try: + yield proc + finally: + # Give pytest a KeyboardInterrupt so its fixtures can clean up the + # separate process groups used by agent commands before pytest exits. + first_signal = signal.SIGINT if interrupt else signal.SIGTERM + with contextlib.suppress(ProcessLookupError): + os.killpg(proc.pid, first_signal) + try: + proc.wait(timeout=15 if interrupt else 5) + except subprocess.TimeoutExpired: + pass + with contextlib.suppress(ProcessLookupError): + os.killpg(proc.pid, signal.SIGKILL) + proc.wait(timeout=5) + + +def exact_npm_version(value: str) -> str: + if not re.fullmatch(r"\d+\.\d+\.\d+(?:-[0-9A-Za-z.-]+)?", value): + raise argparse.ArgumentTypeError( + "Use an exact version, for example 2.1.268; not latest/^/~." + ) + return value + + +def arguments(): + parser = argparse.ArgumentParser(description=__doc__) + source = parser.add_mutually_exclusive_group() + source.add_argument( + "--ug-version", default="checkout", help="Exact ucode release, or checkout." + ) + source.add_argument( + "--ug-wheel", type=Path, help="Previously built wheel to reproduce a release." + ) + parser.add_argument("--entry-point", choices=["ug", "ucode"], default="ug") + parser.add_argument("--claude-version", type=exact_npm_version) + parser.add_argument("--codex-version", type=exact_npm_version) + parser.add_argument("--claude-model", default=os.environ.get("UG_INTEGRATION_CLAUDE_MODEL")) + parser.add_argument("--codex-model", default=os.environ.get("UG_INTEGRATION_CODEX_MODEL")) + parser.add_argument( + "--claude-provider", + default="main.ucode.ci_e2e_anthropic_nonrelay_mps", + help="Existing Anthropic MPS selected in the configure CUJ.", + ) + parser.add_argument( + "--codex-provider", + default="main.ucode.ci_openai_mps", + help="Existing OpenAI MPS selected in the configure CUJ.", + ) + parser.add_argument("--python", default=sys.executable, help="Python 3.12+ path or uv version.") + parser.add_argument("--dependency", action="append", default=[], metavar="PACKAGE==VERSION") + parser.add_argument("--constraints", type=Path, help="Replay a previous dependencies.txt.") + parser.add_argument( + "--npm-lock", type=Path, help="Replay a previous npm-lock.json with npm ci." + ) + parser.add_argument( + "--index-url", default=os.environ.get("UV_INDEX_URL", "https://pypi.org/simple") + ) + parser.add_argument("--npm-registry", default="https://registry.npmjs.org") + parser.add_argument("--profile", help="Explicit Databricks profile to mint the live bearer.") + parser.add_argument("--workspace", default=os.environ.get("UCODE_TEST_WORKSPACE")) + parser.add_argument("--output", type=Path, help="New results directory; never reused.") + parser.add_argument("--installation-only", action="store_true", help="No workspace calls.") + parser.add_argument( + "pytest_args", nargs=argparse.REMAINDER, help="After --, pass pytest filters." + ) + args = parser.parse_args() + # Only selection/early-stop controls are accepted. Pytest configuration, + # plugins and report destinations are part of the suite's isolation contract. + filters = argparse.ArgumentParser(add_help=False) + filters.add_argument("-k") + filters.add_argument("-m") + filters.add_argument("-x", action="store_true") + filters.add_argument("--maxfail", type=int) + extra = args.pytest_args[1:] if args.pytest_args[:1] == ["--"] else args.pytest_args + selected = filters.parse_args(extra) + marker = selected.m or "live" + if args.installation_only: + marker = f"installation and ({selected.m})" if selected.m else "installation" + args.pytest_args = [] + for flag, value in (("-k", selected.k), ("-m", marker), ("--maxfail", selected.maxfail)): + if value is not None: + args.pytest_args.extend([flag, str(value)]) + if selected.x: + args.pytest_args.append("-x") + if not (args.claude_version or args.codex_version): + parser.error("Select --claude-version and/or --codex-version explicitly.") + if args.ug_version != "checkout" and not re.fullmatch( + r"[0-9][0-9A-Za-z.!+_-]*", args.ug_version + ): + parser.error( + "--ug-version must be an exact release, or checkout; use --ug-wheel for a file." + ) + for dependency in args.dependency: + if not re.fullmatch(r"[A-Za-z0-9_.-]+==[A-Za-z0-9_.!+-]+", dependency): + parser.error("--dependency requires an exact PACKAGE==VERSION constraint.") + if not args.installation_only: + if not args.workspace or not args.workspace.startswith("https://"): + parser.error( + "Set UCODE_TEST_WORKSPACE to the existing e2e workspace, or use --workspace." + ) + if not (args.profile or os.environ.get("DATABRICKS_BEARER", "").strip()): + parser.error("Provide the e2e DATABRICKS_BEARER or select --profile explicitly.") + return args + + +def main() -> int: + args = arguments() + if os.name != "posix": + raise SystemExit("This runner supports Linux and macOS. Use the container on other hosts.") + + def terminate(signum, frame): + raise KeyboardInterrupt + + signal.signal(signal.SIGTERM, terminate) + binaries = {name: shutil.which(name) for name in ("uv", "npm", "node", "databricks")} + required = ["uv", "npm", "node"] + ([] if args.installation_only else ["databricks"]) + missing = [name for name in required if not binaries[name]] + if missing: + raise SystemExit("Install these prerequisites first: " + ", ".join(missing)) + if not args.installation_only: + managed_paths = [] + if args.codex_version: + managed_paths.extend( + [Path("/etc/codex/managed_config.toml"), Path("/etc/codex/requirements.toml")] + ) + if args.claude_version: + managed_paths.append( + Path( + "/Library/Application Support/ClaudeCode/managed-settings.json" + if sys.platform == "darwin" + else "/etc/claude-code/managed-settings.json" + ) + ) + present = [str(path) for path in managed_paths if path.exists()] + if present: + raise SystemExit( + "Machine-wide agent settings can override the selected test workspace. " + "Use the integration container instead of this host: " + ", ".join(present) + ) + + stamp = dt.datetime.now(dt.UTC).strftime("%Y%m%dT%H%M%S.%fZ") + output = (args.output or ROOT / ".integration-runs" / stamp).resolve() + output.mkdir(parents=True, exist_ok=False) + build_log = output / "install.log" + # Do not inherit project environments, resolver constraints, pytest options, + # Python optimization, agent credentials, or npm settings from the caller. + keep = ( + "PATH", + "LANG", + "LC_ALL", + "SSL_CERT_FILE", + "SSL_CERT_DIR", + "REQUESTS_CA_BUNDLE", + "NODE_EXTRA_CA_CERTS", + "HTTPS_PROXY", + "HTTP_PROXY", + "NO_PROXY", + ) + base_env = {key: os.environ[key] for key in keep if key in os.environ} + build_home = output / "build-home" + build_home.mkdir() + base_env["HOME"] = str(build_home) + base_env["USERPROFILE"] = str(build_home) + base_env["npm_config_cache"] = str(output / "npm-cache") + base_env["npm_config_fetch_retries"] = "1" + base_env["npm_config_fetch_timeout"] = "30000" + base_env["UV_CACHE_DIR"] = str(output / "cache") + base_env["UV_INDEX_URL"] = args.index_url + bearer = os.environ.get("DATABRICKS_BEARER", "").strip() + + def redact(value: str) -> str: + return value.replace(bearer, "") if bearer else value + + def run(command, *, cwd=output, env=base_env, timeout=600) -> str: + timed_out = False + with managed_process( + [str(x) for x in command], + cwd=cwd, + env=env, + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + ) as proc: + try: + stdout, stderr = proc.communicate(timeout=timeout) + except subprocess.TimeoutExpired: + timed_out = True + if timed_out: + stdout, stderr = proc.communicate(timeout=5) + with build_log.open("a") as log: + log.write(redact(stdout + stderr)) + if timed_out: + raise RuntimeError(f"{command[0]} exceeded {timeout}s; see {build_log}") + if proc.returncode: + raise RuntimeError(f"{command[0]} failed; see {build_log}\n" + redact(stderr[-2000:])) + return stdout.strip() + + report = { + "requested": { + "ug": str(args.ug_wheel) if args.ug_wheel else args.ug_version, + "entry_point": args.entry_point, + "claude": args.claude_version, + "codex": args.codex_version, + "claude_model": args.claude_model, + "codex_model": args.codex_model, + "claude_provider": args.claude_provider, + "codex_provider": args.codex_provider, + "dependencies": args.dependency, + "workspace": args.workspace, + }, + "platform": platform.platform(), + "installation_only": args.installation_only, + } + manifest = output / "versions.json" + exitcode = 1 + try: + print(f"Installing selected versions. Results: {output}", flush=True) + uv = binaries["uv"] + runtime, testenv = output / "ug-runtime", output / "test-runtime" + for path in (runtime, testenv): + run([uv, "venv", "--python", args.python, path]) + python = runtime / "bin/python" + report["python"] = run([python, "--version"]) + report["uv"] = run([uv, "--version"]) + report["node"] = run([binaries["node"], "--version"]) + report["npm"] = run([binaries["npm"], "--version"]) + if binaries["databricks"]: + report["databricks"] = run([binaries["databricks"], "--version"]) + + if args.ug_wheel: + wheel = args.ug_wheel.resolve() + if not wheel.is_file() or wheel.suffix != ".whl": + raise RuntimeError(f"Wheel does not exist: {wheel}") + elif args.ug_version == "checkout": + if not (ROOT / "pyproject.toml").is_file(): + raise RuntimeError( + "No checkout in this image. Pass --ug-version or mount --ug-wheel." + ) + wheels = output / "wheels" + run([uv, "build", "--wheel", "--out-dir", wheels, ROOT], cwd=ROOT) + (wheel,) = wheels.glob("*.whl") + report["git_commit"] = run(["git", "rev-parse", "HEAD"], cwd=ROOT) + report["tracked_diff"] = run( + ["git", "diff", "HEAD", "--", "src", "pyproject.toml"], cwd=ROOT + ) + else: + wheel = None + + if wheel: + report["wheel_sha256"] = hashlib.sha256(wheel.read_bytes()).hexdigest() + package = str(wheel) if wheel else f"ucode=={args.ug_version}" + constraints = output / "requested-constraints.txt" + constraints.write_text( + (args.constraints.read_text() if args.constraints else "") + + "\n" + + "\n".join(args.dependency) + ) + run( + [ + uv, + "pip", + "install", + "--python", + python, + "--index-url", + args.index_url, + "--constraint", + constraints, + package, + ] + ) + run([uv, "pip", "check", "--python", python]) + freeze = run([uv, "pip", "freeze", "--python", python]) + (output / "installed.txt").write_text(freeze + "\n") + (output / "dependencies.txt").write_text( + "\n".join( + line for line in freeze.splitlines() if not re.match(r"ucode(?:==|\s*@)", line) + ) + + "\n" + ) + # The report proves we imported site-packages, not src/ via an editable install. + report["package"] = json.loads( + run( + [ + python, + "-c", + ( + "import importlib.metadata as m, json, ucode; " + "print(json.dumps({'version': m.version('ucode'), 'path': ucode.__file__}))" + ), + ] + ) + ) + package_path = Path(report["package"]["path"]).resolve() + if not package_path.is_relative_to(runtime): + raise RuntimeError( + f"Application was imported outside its isolated environment: {package_path}" + ) + binary = runtime / "bin" / args.entry_point + if not binary.is_file(): + raise RuntimeError( + f"Selected release has no {args.entry_point} entry point; try --entry-point ucode." + ) + + agents = [agent for agent in AGENT_PACKAGES if getattr(args, f"{agent}_version")] + npm_prefix = output / "agents" + npm_prefix.mkdir() + if args.npm_lock: + (npm_prefix / "package.json").write_text( + json.dumps( + { + "dependencies": { + AGENT_PACKAGES[a]: getattr(args, f"{a}_version") for a in agents + } + } + ) + ) + shutil.copyfile(args.npm_lock, npm_prefix / "package-lock.json") + run( + [ + binaries["npm"], + "ci", + "--prefix", + npm_prefix, + "--no-audit", + "--no-fund", + "--registry", + args.npm_registry, + ] + ) + else: + run( + [ + binaries["npm"], + "install", + "--prefix", + npm_prefix, + "--no-audit", + "--no-fund", + "--save-exact", + "--registry", + args.npm_registry, + *[f"{AGENT_PACKAGES[a]}@{getattr(args, f'{a}_version')}" for a in agents], + ] + ) + shutil.copyfile(npm_prefix / "package-lock.json", output / "npm-lock.json") + report["npm_packages"] = json.loads( + run( + [ + binaries["npm"], + "ls", + "--prefix", + npm_prefix, + "--depth=0", + "--json", + ] + ) + )["dependencies"] + agent_bin = npm_prefix / "node_modules/.bin" + # Expose only selected executables to tested programs; other installed + # developer agents cannot be discovered accidentally via inherited PATH. + tool_bin = output / "tools" + tool_bin.mkdir() + for name in ("node", "databricks"): + if binaries[name]: + (tool_bin / name).symlink_to(binaries[name]) + runtime_env = dict(base_env) + runtime_env["PATH"] = os.pathsep.join( + map(str, [runtime / "bin", agent_bin, tool_bin, "/usr/bin", "/bin"]) + ) + report["agents"] = {} + for agent in agents: + version = run([agent_bin / agent, "--version"], env=runtime_env, timeout=30) + expected = getattr(args, f"{agent}_version") + if not re.search(rf"(? dict: - backup_existing_file(CLAUDE_SETTINGS_PATH, CLAUDE_BACKUP_PATH) + # Back up only a file that predates ucode's management of the tool. A + # re-configure would otherwise snapshot ucode's own generated file, and + # revert would restore that snapshot instead of deleting the file. + if not is_tool_managed(state, "claude"): + backup_existing_file(CLAUDE_SETTINGS_PATH, CLAUDE_BACKUP_PATH) web_search_model = _resolve_web_search_model(state) # Relayed inference points at a local refresh proxy; its loopback base URL is # recorded in state so launch starts the proxy on the matching port. diff --git a/src/ucode/agents/codex.py b/src/ucode/agents/codex.py index 09de63612..5e69d849d 100644 --- a/src/ucode/agents/codex.py +++ b/src/ucode/agents/codex.py @@ -63,7 +63,7 @@ sync_smart_routing_hooks, ) from ucode.smart_routing.codex_routing import codex_model_id -from ucode.state import get_provider_service, mark_tool_managed, save_state +from ucode.state import get_provider_service, is_tool_managed, mark_tool_managed, save_state from ucode.telemetry import agent_version, ucode_version from ucode.ui import print_warning_err @@ -399,7 +399,11 @@ def write_tool_config( return state _remove_legacy_ucode_profile() - backup_existing_file(CODEX_CONFIG_PATH, CODEX_BACKUP_PATH) + # Back up only a file that predates ucode's management of the tool. A + # re-configure would otherwise snapshot ucode's own generated file, and + # revert would restore that snapshot instead of deleting the file. + if not is_tool_managed(state, "codex"): + backup_existing_file(CODEX_CONFIG_PATH, CODEX_BACKUP_PATH) overlay = render_overlay( workspace, chosen_model, @@ -572,7 +576,10 @@ def clear_model_preferences(state: dict) -> bool: doc.pop(key) changed = True if changed: - backup_existing_file(CODEX_CONFIG_PATH, CODEX_BACKUP_PATH) + # Never snapshot ucode's own generated file here; revert would restore + # the snapshot instead of deleting the file. + if not is_tool_managed(state, "codex"): + backup_existing_file(CODEX_CONFIG_PATH, CODEX_BACKUP_PATH) write_toml_file(CODEX_CONFIG_PATH, doc) return changed diff --git a/src/ucode/cli.py b/src/ucode/cli.py index 9a04223b5..679954928 100644 --- a/src/ucode/cli.py +++ b/src/ucode/cli.py @@ -140,6 +140,7 @@ prompt_for_tools, prompt_for_workspace, prompt_yes_no, + redirect_output_to_stderr, set_verbosity, spinner, status_badge, @@ -1986,6 +1987,16 @@ def _download_managed_skills(managed: dict, state: dict) -> None: print_note(f"Downloaded workspace skill(s) to disk: {', '.join(written)}") +def _child_owns_stdout(tool: str, tool_args: list[str]) -> bool: + """True when the forwarded agent command speaks a stdio protocol on stdout. + + ``codex app-server`` puts its JSON-RPC stream on stdout, so ug's status + output must move to stderr for that launch; the file descriptor stays + untouched for the agent process. + """ + return tool == "codex" and tool_args[:1] == ["app-server"] + + def _should_launch_smart_routing( tool: str, tool_args: list[str], @@ -2049,6 +2060,10 @@ def _launch_tool( ) -> None: try: tool = normalize_tool(tool_name) + # Before any status print: a stdio-protocol subcommand owns stdout, so + # every ug line from here on must go to stderr instead. + if _child_owns_stdout(tool, ctx.args): + redirect_output_to_stderr() if provider is not None and parent_schema is not None: raise RuntimeError("--provider and --parent cannot be used together.") if parent_schema is not None and not is_valid_catalog_schema(parent_schema): diff --git a/src/ucode/state.py b/src/ucode/state.py index c02855537..e37f60e3e 100644 --- a/src/ucode/state.py +++ b/src/ucode/state.py @@ -251,6 +251,16 @@ def clear_state() -> None: raise RuntimeError(f"Failed to clear state file: {STATE_PATH}") from exc +def is_tool_managed(state: dict, tool: str) -> bool: + """True once configure has written ucode's config for ``tool``. + + Used to distinguish a config file the user owned before ucode ran from one + ucode itself generated: only the former is backup-worthy, because revert + restores a backup in place instead of deleting ucode's generated file. + """ + return bool((state.get("managed_configs") or {}).get(tool)) + + def mark_tool_managed(state: dict, tool: str, managed_keys: list) -> dict: """Record which config keys ucode manages for ``tool``.""" managed_configs = dict(state.get("managed_configs") or {}) diff --git a/src/ucode/ui.py b/src/ucode/ui.py index ce92bcbcf..c2c4af57c 100644 --- a/src/ucode/ui.py +++ b/src/ucode/ui.py @@ -24,6 +24,19 @@ console = Console(highlight=False) err_console = Console(stderr=True, highlight=False) + +def redirect_output_to_stderr() -> None: + """Move status output to stderr because a child process now owns stdout. + + ``codex app-server`` speaks its JSON-RPC protocol on stdout, so any ug line + printed there corrupts the stream. Rich resolves ``console.file`` to + ``sys.stdout`` at print time, so rebinding the module attribute sends every + status print to stderr while ``os.execvp`` still hands the launched agent + the untouched stdout file descriptor. + """ + sys.stdout = sys.stderr + + # Past this many options the choice list is pinned to a fixed-height scrolling viewport (see # `_cap_choice_viewport`) rather than growing to fill the terminal, and the pickers append a # "↑/↓ scroll" note to their instruction line. The value is both the boundary and the number of diff --git a/tests/AGENTS.md b/tests/AGENTS.md new file mode 100644 index 000000000..7d20b5dee --- /dev/null +++ b/tests/AGENTS.md @@ -0,0 +1,113 @@ +# Instructions for agents editing tests + +Read `README.md` and `integration/README.md` before adding, removing, or changing +tests. Keep work scoped to the behavior requested by the user. + +## Categories + +- Existing unit/component tests may use focused mocks. Do not move their global + fixtures into integration or rewrite them all as part of a narrow fix. +- `integration/` tests installed ug with real agents and the real + `UCODE_TEST_WORKSPACE` already used by e2e. The following rules cover every + test, helper, and fixture in that directory. + +## Integration rules: no test hacks + +1. **No mocks or monkeypatching.** No `monkeypatch`, `pytest.MonkeyPatch`, + `unittest.mock`, `Mock`, `MagicMock`, `patch`, replacement executables, fake + HTTP services, import substitution, or runtime alteration of application code. +2. **Do not import application internals.** Exercise installed `ug` / `ucode` + and agent commands through subprocesses. Do not call config writers, + launchers, parsers, authentication helpers, or state helpers directly. +3. **Create ug state through public CLI commands.** Never hand-write ug state, + cached discovery, or generated gateway config to get past setup. Ordinary + input files and pre-existing user-owned settings are valid scenario inputs; + identify them clearly and assert their preservation. +4. **No production changes just to make tests pass.** No test-only environment + switches, special server branches, disabled validation, privileged-path + overrides, or hardcoded success. Real bugs require normal production fixes + and regression coverage. Report failures instead of concealing them. +5. **Real responses and binaries.** Pin requested ug and agent versions. Never + substitute a missing binary/service. Reuse explicit e2e workspace/auth settings; + never pick a developer's Databricks profile automatically. +6. **Fail honestly.** Missing prerequisites/capabilities, timeouts, protocol errors, + and unexpected nonzero exits fail. Do not add skips, xfails, broad exception + suppression, task retries, or weaker assertions to make CI green. Report + deliberately selected subsets explicitly. +7. **Assert observable behavior.** Use actual files, structured final agent + results, exit status and protocol exchanges. Process startup, banners, echoed + input, or nonempty output alone do not establish a successful task. +8. **Isolate through processes and environments.** Fresh homes, working directories, + package environments and explicit subprocess `env` are allowed. Do not mutate + pytest's environment to redirect imported modules. Avoid developer config, + credentials, OS-managed paths and installed tools. +9. **Bound and clean up work.** Give every process a timeout and explicit stdin; + reap children on failure. Do not swallow timeouts or leave servers running. + Keep live prompts small. Record models used by explicit-model scenarios; + normal boot should use the workspace's own configuration. +10. **Evidence without secrets.** Record versions, dependencies, command shape, + exit codes and redacted diagnostics. Never archive tokens, credential files, + complete environments or developer homes. +11. **Drive the real TUI.** PTYs and terminal screen parsers are allowed. Handle + onboarding and trust through visible UI choices and actual keystrokes; never + seed onboarding completion, intercept model traffic, or replace a TUI with + print/exec mode while claiming interactive coverage. Require an interactive + prompt and observable input/exit behavior. Boot is not first-prompt inference. + +`configure --skip-validate` in setup avoids an extra generic model prompt; task +tests then require a real independently asserted task. Configuration alone must +never be presented as inference coverage. Do not add shortcuts around the +behavior a test claims to exercise. + +## Add / modify / remove + +### Required integration test format + +- Organize the suite around complete user journeys, with explicit names + such as `test_ug_configure_claude_databricks` or + `test_smart_routing_codex_first_prompt`. Do not hide the agent/provider behind + generic parametrization in these tests. +- Every test has a docstring with **Scenario:** and **Expected:**. State what the + user does and the observable evidence required for success, including limits. +- Keep the configure command, launch, task, and assertions visible in the test. + Fixtures provide fresh environments and credentials, never a preconfigured app. +- Helpers may handle processes, terminal keys, transcript parsing, cleanup, and + artifact collection. Do not bury an entire CUJ inside an opaque helper. +- Provider configuration journeys must complete a real TUI task. A startup banner, + config file, echoed prompt, or tool output alone does not prove completion. +- Keep all CUJs as descriptive top-level `integration/test_*.py` files. Do not + create a separate regressions category. Shared process/terminal/evidence helpers + and Docker build files belong in `integration/utils/`; keep pytest entry points + and run documentation at the suite root. +- Current scope is basic Claude/Codex configuration, routing, script usage, + command forwarding, and configure/revert CUJs. Do not add MCP/skills functionality, + tracing, or the broad configure-option matrix without a new scope request. + Existing `mcp --help` checks cover dispatch only. + +- **Add:** state the user scenario and affected versions, choose the category, + add a focused test and coverage row. Demonstrate regression failure on the + affected combination when it is available. +- **Modify:** preserve or strengthen assertions. Explain a changed product + contract rather than silently redefining success. Update both coverage READMEs + and scenario/version inputs when their claims change. +- **Remove:** explain obsolete/duplicate coverage and where any still-required + behavior is tested. Mark remaining gaps as not covered. Never remove a case + merely because a real agent or gateway currently fails it. +- Help is not desktop startup; configuration is not a completed task; routing + bypass is not successful routing; launcher-style arguments are not Isaac. + +## Verification + +```bash +uv run pytest tests/test_integration_contract.py +uv run ruff check tests/ scripts/run_integration.py +uv run ruff format --check tests/ scripts/run_integration.py +python3 scripts/run_integration.py --help +``` + +Run integration through `scripts/run_integration.py` with explicit versions. +Use `--installation-only` for package checks and `-- -k EXPRESSION` for a reported +subset. Live checks require the e2e workspace and bearer/profile. Model overrides +are optional inputs for reproducing an explicit-model failure. +Report passes, failures and what was not run. Collection/lint/package checks +are not evidence of a live integration pass. diff --git a/tests/CLAUDE.md b/tests/CLAUDE.md new file mode 100644 index 000000000..de5260a4c --- /dev/null +++ b/tests/CLAUDE.md @@ -0,0 +1,12 @@ +# Test instructions for Claude Code + +Read and follow [AGENTS.md](AGENTS.md) in this directory and the coverage matrix +in [README.md](README.md) before changing tests. Those are the shared instructions +for adding, modifying and removing tests; do not maintain a different policy here. + +In particular, `integration/` uses the real installed ug CLI, selected real agent +versions, and the real e2e workspace. No mocks, monkeypatching, internal application +calls, fake binaries/services, fabricated ug state, or test-only production +behavior. Do not hide failures with skips, xfails, retries or weaker assertions. +Update the matrix when coverage changes and distinguish automated coverage from +gaps and checks that were not executed. diff --git a/tests/README.md b/tests/README.md new file mode 100644 index 000000000..95b886fc9 --- /dev/null +++ b/tests/README.md @@ -0,0 +1,83 @@ +# Test suites and user journeys + +Integration runs a freshly installed ug wheel/release, exact real Claude/Codex +versions, and the existing real e2e workspace. It has no application imports, +mocks, monkeypatching, fake binaries/services, or fabricated ug state. + +| Category | Location | What it proves | +| --- | --- | --- | +| Unit/component | Existing `test_*.py` files | Individual behavior; dependencies may be mocked | +| Existing e2e | `test_e2e*.py` | Real workspace behavior with some patched setup/internal calls | +| Integration CUJs | `integration/test_*.py` | Public configure, TUI, script, command, protocol, and lifecycle journeys | +| Installation | `integration/test_installation.py` | Fresh installed package without credentials | + +## CUJ coverage matrix + +These are **implemented assertions**, not a claim that every version passes. +Consult the run's JUnit report and artifacts for results. Each function states +its **Scenario** and **Expected** outcome and shows its configure and launch +commands. Fixtures supply fresh environments and credentials, never configured ug. +All tests live directly in `integration/`; shared mechanics live in `utils/`. + +| Test | User action | Expected evidence | +| --- | --- | --- | +| `test_ug_configure_claude_databricks` | Configure Databricks Hosted; open Claude TUI and read a file | Assistant returns an unpredictable file value; normal exit; reopen with working keyboard input | +| `test_ug_configure_claude_anthropic_mps` | Select Anthropic MPS in the real configure picker; launch Claude | Saved provider in status; completed TUI file task; normal exit | +| `test_ug_configure_codex_databricks` | Configure Databricks Hosted; open Codex TUI and read a file | Completed assistant answer contains the file value; normal exit and reopen | +| `test_ug_configure_codex_openai_mps` | Select OpenAI MPS in the real configure picker; launch Codex | Saved provider in status; completed TUI file task; normal exit | +| `test_smart_routing_claude_first_prompt` | Enable routing and type the first TUI prompt | Real routing decision and replay; completed task; no routing before submission; reopen | +| `test_smart_routing_codex_first_prompt` | Enable routing and type the first TUI prompt | Real routing decision; completed task; no fallback or routing before submission; reopen | +| `test_smart_routing_claude_subagent` | Delegate a file-reading task | Child answer, correlated routing decision/child start, parent answer | +| `test_smart_routing_codex_subagent` | Delegate a file-reading task | Native child-parent linkage; completed child turn uses the routed model; parent answer | +| `test_smart_routing_claude_explicit_model_bypasses_routing` | Launch TUI with an explicit model and routing enabled | Completed task, no routing wrapper, normal exit and reopen | +| `test_smart_routing_codex_explicit_model_bypasses_routing` | Launch TUI with an explicit model and routing enabled | Completed task, no routing wrapper, normal exit and reopen | +| `test_ug_claude_headless_prompt_argument`, `test_ug_claude_headless_prompt_stdin`, `test_ug_claude_headless_prompt_after_separator` | Run Claude from a script using each prompt form | Structured final answer contains the file value; exit zero; no routing | +| `test_ug_codex_headless_prompt_argument`, `test_ug_codex_headless_prompt_stdin`, `test_ug_codex_headless_prompt_after_separator` | Run Codex from a script using each prompt form | Completed turn and final answer contain the file value; exit zero; no routing | +| `test_ug_claude_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` with routing enabled | Real file task completes; no routing wrapper | +| `test_ug_codex_headless_explicit_model_bypasses_routing` | Pass `--model VALUE` / `--model=VALUE` / `-m VALUE` with routing enabled | Real file task completes; no routing wrapper | +| `test_ug_claude_preserves_caller_settings_and_hook` | Pass a settings path containing spaces | Real SessionStart hook executes; caller file unchanged; file task completes | +| `test_ug_claude_reports_unsupported_short_model_option` | Pass Claude's unsupported `-m` | Actual agent error and exit status preserved | +| `test_ug_claude_auth_help`, `test_ug_claude_mcp_help` | Request subcommand help, routing off/on | Real agent help; no routing wrapper | +| `test_ug_codex_app_help`, `test_ug_codex_app_server_help`, `test_ug_codex_exec_help`, `test_ug_codex_mcp_help` | Request subcommand help, routing off/on | Real agent help; no routing wrapper | +| `test_ug_codex_app_reports_unknown_argument` | Pass an invalid option directly to `ug codex app`, routing off/on | Real Codex parser error and status preserved | +| `test_ug_codex_app_server_client_initializes` | Connect a stdio client, direct/`--` separator, routing off/on | Actual JSON-RPC initialize response; no non-JSON stdout; no routing | +| `test_ug_configure_claude_repeat_and_revert`, `test_ug_configure_codex_repeat_and_revert` | Configure twice over user settings; complete a task; revert twice | Settings preserved; no bearer in ug state; generated config removed; status unconfigured | +| `test_ug_configure_claude_rejects_invalid_credentials`, `test_ug_configure_codex_rejects_invalid_credentials` | Configure with a rejected bearer against the real workspace | Authentication failure; no successful saved setup | +| `test_ug_installed_wheel_exposes_help_and_version` | Invoke freshly installed console command | Package version matches; public help works | +| `test_ug_status_in_fresh_home_is_unconfigured` | Request status before configure | Unconfigured status | +| `test_ug_auth_without_configuration_explains_how_to_configure` | Request auth before configure | Actionable setup error and nonzero exit | + +With both agents selected there are **45 live cases** (10 interactive TUI cases) +and **3 installation checks**. Parametrization varies argument spelling or routing +mode, never hides the agent/provider in the test name. Duplicate boot-only cases +were merged into the Databricks, first-prompt, and explicit-model TUI journeys. +Generated-file cleanup and strict app-server stdout assertions remain enforced. + +Provider configuration tests include ug's normal validation. Other setup uses +`--skip-validate` when the journey supplies its own task or checks a command +contract. Tests use `--skip-upgrade` to preserve selected agent versions and disable +optional Databricks AI Tools. Help forwarding does not claim MCP functionality. + +Fresh consumer dependency resolution covers the install path behind #496, rather +than consuming `uv.lock`. Use `--dependency PACKAGE==VERSION` or replay the archived +dependency graph to reproduce a user's combination. PR CI runs one live CUJ job. + +## Gaps and deferred scope + +| Scenario | Status / requirement | +| --- | --- | +| MCP and skills functionality | Deferred at the user's request; existing `mcp --help` dispatch checks only | +| Broad configure flags, tracing, multiple workspaces, OAuth/PAT flows | Deferred while focusing on basic CUJs | +| Provider switching, relayed/subscription MPS | Not covered by the four provider journeys | +| TUI initial prompt supplied on the launch command line | Not yet covered; headless prompt arguments are covered | +| Follow-up turns and conversation resume | Not covered; reopen proves startup, not conversation resume | +| Claude child model identity | Verified only when the actual start event reports it; missing fields remain unknown | +| Full allow/deny tool-permission matrix | Not covered; onboarding/trust uses actual TUI choices | +| Desktop Codex app, Isaac itself, auto-upgrades | Not covered by command forwarding or pinned-version tests | +| Native macOS/Windows managed settings, resize/signals | Separate platform coverage needed | +| Other agents | Current scope is Claude Code and Codex | + +See [integration/README.md](integration/README.md) for commands, CI, artifacts, +and reproduction. Follow [AGENTS.md](AGENTS.md) and [CLAUDE.md](CLAUDE.md) when +adding, modifying, or removing tests. The ordinary suite enforces both the +no-mocking boundary and the Scenario/Expected docstring format. diff --git a/tests/conftest.py b/tests/conftest.py index 737972b87..fde1c180b 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -15,6 +15,10 @@ ) from ucode.ui import normalize_workspace_url +# The integration suite has its own configuration and subprocess-only fixtures. +# Run it through scripts/run_integration.py, outside this fixture hierarchy. +collect_ignore = ["integration"] + @pytest.fixture(autouse=True) def _isolate_ucode_state(tmp_path, monkeypatch): diff --git a/tests/integration/README.md b/tests/integration/README.md new file mode 100644 index 000000000..cf0c2851e --- /dev/null +++ b/tests/integration/README.md @@ -0,0 +1,303 @@ +# Integration tests + +This suite runs the **installed product** through subprocesses, against the same +`UCODE_TEST_WORKSPACE` used by the existing e2e tests. It does not import `ucode`, +patch application functions, substitute agent executables, run a fake gateway, +or construct ug state files. The normal test suite checks these boundaries. + +The existing unit tests keep their fixtures. Integration has an independent +pytest configuration and uses `--confcutdir` so those fixtures cannot leak in. +It is not collected by the default `uv run pytest` command. + +## Run a specific combination + +Prerequisites: Python 3.12+, uv, Node/npm, and Databricks CLI >=1.0.0. The runner +installs the requested agents into a new npm prefix and ug into a new virtualenv. +Pytest and the PTY/screen libraries (pexpect and pyte) live in a different virtualenv, so they cannot accidentally supply a +missing application dependency. No packages are installed into your existing +agent installations or checkout's `.venv`. +Native live runs refuse existing machine-wide Claude/Codex configuration, which +could override the selected workspace even with a fresh home. Use the container +in that case; the runner never edits or bypasses those managed settings. + +Use the existing e2e workspace and its `DATABRICKS_BEARER` credential. Locally, +`--profile YOUR_PROFILE` can mint a bearer for an explicitly selected profile. +No profile or workspace is selected automatically. The default test selection is `live` (all live CUJs). + +```bash +export UCODE_TEST_WORKSPACE=https://your-existing-e2e-workspace + +python3 scripts/run_integration.py \ + --ug-version checkout \ + --claude-version 2.1.268 \ + --codex-version 0.154.0 \ + --profile YOUR_PROFILE +``` + +`checkout` builds a wheel and installs it with fresh consumer dependency +resolution. **It does not use `uv.lock`.** This exercises the install path that +caught the tomlkit discrepancy in #496. To reproduce a user's release, pass its +exact distribution version instead, e.g. `--ug-version 0.1.0+f7b4b97`. Use +`--index-url` for the Python index that contains that release and `--npm-registry` +for an npm mirror if public npm is unavailable. Older releases that only +provide the `ucode` command require `--entry-point ucode`. + +Select one agent by providing only its version. Exact agent versions are +required; floating `latest`, caret, and tilde versions are rejected. Provider and default-routing CUJs use the workspace's configuration and need no model input. Cases that +exercise explicit model arguments use a real `system.ai` model already discovered +by `ug configure`, recorded in that case's `model.json`. Optional `--claude-model` +and `--codex-model` overrides reproduce a particular model-related failure. + +```bash +# Constrain the suspected dependency while keeping the real CLI and gateway. +python3 scripts/run_integration.py \ + --ug-version checkout --claude-version 2.1.268 --codex-version 0.154.0 \ + --claude-model YOUR_CLAUDE_MODEL --codex-model YOUR_CODEX_MODEL \ + --profile YOUR_PROFILE --dependency tomlkit==0.14.0 \ + -- -k 'app_server or app_help' +``` + +Repeat with `--dependency tomlkit==0.15.1`, or run without constraints to test +what a new consumer gets today. Several `--dependency` options can be supplied. +Constraints incompatible with the selected ug release fail installation. + +For package-only validation without credentials: + +```bash +python3 scripts/run_integration.py \ + --ug-version checkout --claude-version 2.1.268 --installation-only +``` + +This explicitly selects only the installation checks; it does not claim a live +integration pass. Requested live checks fail when credentials, binaries, models, +or capabilities are missing. There are no capability-based skips or retries of +failed model tasks. A failing historical version should remain a failing result. + +## Test layout and format + +All user journeys are top-level tests. There is no separate regressions category: + +```text +test_ug_configure_claude.py # Databricks Hosted and Anthropic MPS +test_ug_configure_codex.py # Databricks Hosted and OpenAI MPS +test_smart_routing_claude.py # first prompt, subagent, explicit model +test_smart_routing_codex.py # first prompt, subagent, explicit model +test_ug_claude_headless.py # script prompts, models, caller settings +test_ug_codex_headless.py # script prompts and model arguments +test_ug_claude_commands.py # command help forwarding +test_ug_codex_commands.py # command help and parser error forwarding +test_ug_codex_app_server.py # actual client/server initialize exchange +test_ug_configure_claude_lifecycle.py # repeat setup, revert, rejected credentials +test_ug_configure_codex_lifecycle.py # repeat setup, revert, rejected credentials +test_installation.py # fresh installed package +utils/ # process/terminal/evidence helpers and Docker files +``` + +`conftest.py`, `pytest.ini`, and this README stay at the suite root for pytest +discovery and run instructions. + +Each test has a `Scenario:` / `Expected:` docstring and shows its own public +configure command, launch, user action, and assertions. Shared code only handles +process/terminal mechanics, evidence, and cleanup. Fixtures supply an isolated +session and credentials; none manufacture or configure application state. + +Provider CUJs use normal ug validation, then require their own completed +interactive task. Routing CUJs skip the preliminary validation prompt and require +the actual routed TUI task instead. Tests disable optional Databricks AI Tools and +pass `--skip-upgrade` to preserve the selected version. They use real onboarding +and trust choices, without seeded acceptance or disabled agent sandboxing. + +A fixture file contains an unpredictable value absent from the prompt. Success +requires an assistant answer in the real agent transcript containing that value, +plus normal TUI exit. Codex evidence requires its task-complete event. Subagent +CUJs require a separate child transcript, child answer, and a correlated routing +decision. Claude uses its real child-start audit; its model is unknown when the +event omits it. Codex requires native parent linkage and a completed child turn +using the routed model, excluding inherited parent turns. + +MPS CUJs select the existing services already used by e2e: + +- Claude: `main.ucode.ci_e2e_anthropic_nonrelay_mps`. +- Codex: `main.ucode.ci_openai_mps`. + +Use `--claude-provider` / `--codex-provider` to reproduce another existing service. +Those names are recorded in `versions.json`. No service is created or modified. +A missing service or permission fails the selected CUJ, rather than skipping it. + +There are **45 live cases** (including 10 TUI journeys) and **3 installation +checks** with both agents. See the named coverage and gaps matrix in +[../README.md](../README.md). + +```bash +# Append one of these selections to the runner command: +-- -m live # default: all live user journeys +-- -m tui # ten complete interactive TUI journeys +-- -k test_ug_codex_app_server_client_initializes # one named journey and its variants +# Use --installation-only before -- for package checks without credentials. +``` + +The old focused checks are now descriptive CUJs with setup and outcomes visible +in each test. Duplicate boot-only checks are incorporated into the Databricks, +first-prompt, and explicit-model TUI journeys. Real failures, including generated +config left after revert and banners on app-server stdout, remain assertions. +MCP/skills functionality, tracing, the broad configure-option matrix, and other +agents are outside this focused revision. + +## Reproduce a failure + +Each run writes a new `.integration-runs//` directory containing: + +- `versions.json`: requested and observed ug/agent versions, Python, Node, uv, + Databricks CLI, platform, source revision/diff, suite hash, and wheel hash when available. +- `dependencies.txt` and `npm-lock.json`: the resolved Python and npm dependency + graphs. Replay them with `--constraints` and `--npm-lock`. +- `test-dependencies.txt`: the separately installed pytest/terminal-tool dependencies. +- `junit.xml`: exact test outcomes and parametrized case names. +- `artifacts/`: command arguments, exit codes, timeout status, redacted output, + app-server protocol diagnostics, and TUI transcripts/rendered screens plus + keystroke actions and routing logs. No credential files are archived. +- `wheels/`: the tested wheel when built from the checkout; replay it with + `--ug-wheel`. For release installations, `installed.txt` records the resolution. + +Teardown invokes real `ug revert` through a PTY when setup created state, restoring machine-level +configuration through the public CLI. Per-test homes and working directories are +then deleted even on failure. +The working directory is outside the checkout so an agent cannot inherit its +project settings or instruction files by walking parent directories. Virtualenvs, +agent packages, and build caches remain under the results directory for local +inspection; remove that run directory when finished. Agent versions are checked +before and after the suite so an automatic upgrade cannot silently change the +combination being tested. Model requests and subprocesses have deadlines, and +the process group is cleaned up after each command. +Selection after `--` accepts `-k`, `-m`, `-x`, and `--maxfail`; configuration and +report paths cannot be overridden. `--installation-only` always restricts the +selection to installation checks, including when additional filters are used. + +## Run in GitHub Actions + +The **Integration** workflow runs on relevant pull requests and pushes to `main`. +It runs directly on fresh GitHub Ubuntu VMs, not inside the optional Docker image. +Local native runs use the same runner; Colima/Docker provides a separate Linux +container option. Matching dependency versions does not make those OS environments identical. +Its installation job needs no credentials. For same-repository PRs, the live job +reuse the existing `UCODE_TEST_WORKSPACE` and `DATABRICKS_BEARER` secrets. Fork PRs +run installation checks only because they cannot receive those secrets. + +The workspace check requires the secret to match +`https://eng-ml-inference-team-us-east-1.cloud.databricks.com` (a trailing slash +is accepted). It never changes the secret or switches workspaces. There is no CI +model-discovery or model-selection job. Real `ug configure` performs its normal +workspace discovery inside each test; only explicit-model scenarios choose and +record a discovered `system.ai` model as a test argument. +PR CI runs all 45 live cases in one **CUJs** job with fresh consumer dependency +resolution. There is no default dependency matrix. Manual dispatch accepts an +optional `dependency` such as `tomlkit==0.14.0`, equivalent to the local runner's +`--dependency` option. Jobs use Ubuntu 22.04; newer Ubuntu runner +policies prevented Codex's bubblewrap tool from reading even the test file in the +first run. The agent sandbox is not disabled or bypassed. +The workflow consumes the stored bearer; it does not mint or refresh credentials. + +For a manual run, use **Actions → Integration → Run workflow**, select the branch, +and choose `live` (default), `tui`, or `installation`. Set the ug/agent versions. From the CLI: + +```bash +gh workflow run integration.yml -R databricks/unity-gateway --ref YOUR_BRANCH \ + -f suite=live -f ug_version=checkout \ + -f claude_version=2.1.268 -f codex_version=0.154.0 +gh run list -R databricks/unity-gateway --workflow integration.yml +gh run watch RUN_ID -R databricks/unity-gateway --exit-status +``` + +GitHub enables manual dispatch once the workflow exists on the default branch. +Before this PR merges, its pull-request event runs the workflow. Missing +credentials or a workspace mismatch fail the workspace job. Expired or invalid +credentials fail the actual workspace calls. Those failures do not count as live +test passes. + +## Reproduce and debug a CI failure locally + +Use the same runner and the failing job's artifacts. A new developer machine +needs Python 3.12+, uv, Node/npm, Databricks CLI, and its own authorized login for +the CI workspace. Select that local profile explicitly; CI secrets are not downloaded. + +```bash +gh run download RUN_ID -R databricks/unity-gateway \ + -n integration-cujs -D .integration-runs/from-ci +``` + +Use `integration-installation` for package failures. Older runs used numbered +`integration-live-*` artifacts; download the name shown on that run. Read `versions.json` for the +exact agent versions, model overrides, entry point, platform and source revision. +For an explicit-model case without a runner override, read its `model.json` for +the exact model used. Basic boot cases require no model arguments. Use +the archived wheel so a changed checkout cannot alter the reproduction: + +```bash +python3 scripts/run_integration.py \ + --ug-wheel .integration-runs/from-ci/wheels/EXACT_WHEEL.whl \ + --entry-point ug \ + --claude-version CLAUDE_VERSION_FROM_REPORT --codex-version CODEX_VERSION_FROM_REPORT \ + --workspace https://eng-ml-inference-team-us-east-1.cloud.databricks.com \ + --profile YOUR_E2E_PROFILE \ + --constraints .integration-runs/from-ci/dependencies.txt \ + --npm-lock .integration-runs/from-ci/npm-lock.json \ + --output .integration-runs/repro-1 \ + -- -k test_ug_configure_claude_databricks +``` + +For an explicit-model failure, also pass the recorded `--claude-model` or +`--codex-model`. For a release run without an archived wheel, use the reported `--ug-version`. +Match Python and Node versions from the report too. `npm-lock.json` replay must +use the same OS/architecture as the original run; add `--platform linux/amd64` +to both `docker build` and `docker run` on an ARM Mac to match GitHub's Ubuntu runner. Changing platforms or +resolving a fresh npm lock is a new comparison, not an exact dependency replay. + +Use `-- -m tui` for the ten interactive journeys or +`-- -k test_smart_routing_codex_first_prompt` to narrow a failure. Each rerun needs a new output directory. Inspect: + +- `junit.xml` for the failing case and assertion. +- `artifacts//command-*.json` for the real argv, exit status, stdout and stderr. +- `artifacts//first-session.json`, `provider-session.json`, `first-prompt.json`, + `subagent-task.json`, or `reopen.json` for rendered terminal screens, raw terminal + output, keyboard actions, exit status, routing logs, and actual agent-session records. +- `install.log` for resolution/bootstrap failures. + +Unknown onboarding screens fail with their actual screen text. Update terminal +selectors only after confirming the agent's intended UI changed; do not seed its +onboarding state or relax the prompt/task assertions. Test homes are deleted after +each case; redacted diagnostics remain. For manual interaction, configure a fresh +home with the same installed binaries and recorded public CLI arguments. + +## Colima / Docker + +Colima provides the Linux Docker engine on macOS. The optional image pins the +Python, Node, uv, and Databricks toolchain; the same runner selects ug and agent +versions inside it. Build from the repository root: + +```bash +colima start +COPYFILE_DISABLE=1 tar --format=ustar --exclude=__pycache__ --exclude=.pytest_cache \ + -cf - scripts/run_integration.py tests/integration | \ + docker build -f tests/integration/utils/Dockerfile -t ug-integration - + +# Reuse the same e2e variables. Credentials are passed at runtime, never built +# into the image. The named volume keeps results after the container exits. +docker volume create ug-integration-results +docker run --rm --init \ + -e UCODE_TEST_WORKSPACE -e DATABRICKS_BEARER \ + -v ug-integration-results:/results \ + ug-integration \ + --ug-version YOUR_RELEASE_VERSION \ + --claude-version 2.1.268 --codex-version 0.154.0 \ + -- -m live +``` + +Use a new results volume for each run, or pass a new `--output /results/NAME`. +To test a checkout, build a wheel on the host (`uv build --wheel`), mount the +wheel directory read-only, and pass `--ug-wheel /wheels/FILE.whl` instead of a +release version. The image deliberately contains no source checkout or host +agent configuration. Record the built image digest when sharing a reproduction; +native runs also depend on the host's OS and toolchain. +The explicit build archive includes only the runner and integration files, even +with legacy Docker builders that ignore per-Dockerfile ignore rules. It also +omits macOS extended attributes that Linux cannot unpack. diff --git a/tests/integration/conftest.py b/tests/integration/conftest.py new file mode 100644 index 000000000..f3ce72b29 --- /dev/null +++ b/tests/integration/conftest.py @@ -0,0 +1,93 @@ +"""Standalone fixtures: application state is created only by installed CLI commands.""" + +from __future__ import annotations + +import os +import re +import shutil +import tempfile +from pathlib import Path + +import pytest +from utils.harness import UserSession +from utils.terminal import TerminalProcess + + +def pytest_collection_modifyitems(config, items): + agents = os.environ.get("UG_INTEGRATION_AGENTS", "claude,codex").split(",") + selected, deselected = [], [] + for item in items: + # Fail if someone accidentally invokes this under the unit-test fixtures. + if "monkeypatch" in item.fixturenames: + raise pytest.UsageError("Use the integration runner; unit fixtures were inherited.") + if any(item.get_closest_marker(a) and a not in agents for a in ("claude", "codex")): + deselected.append(item) + else: + selected.append(item) + items[:] = selected + config.hook.pytest_deselected(items=deselected) + + +@pytest.fixture(scope="session") +def installed_binary(): + raw = os.environ.get("UG_INTEGRATION_BIN") + if not raw or not Path(raw).is_file(): + pytest.fail("Run scripts/run_integration.py to install the version under test.") + return Path(raw) + + +@pytest.fixture(scope="session") +def workspace(): + value = os.environ.get("UCODE_TEST_WORKSPACE", "").strip().rstrip("/") + if not value.startswith("https://") or not os.environ.get("DATABRICKS_BEARER", "").strip(): + pytest.fail("Live integration requires UCODE_TEST_WORKSPACE and DATABRICKS_BEARER.") + return value + + +@pytest.fixture +def session(request, installed_binary): + # Codex rejects helper installation beneath /tmp. Keep the disposable home + # under the runner's own directory, never in the developer's agent folders. + root = Path(os.environ["UG_INTEGRATION_RUN_DIR"]) + case = re.sub(r"[^a-zA-Z0-9_.-]", "_", request.node.name) + with ( + tempfile.TemporaryDirectory(prefix="case-", dir=root) as temporary, + tempfile.TemporaryDirectory(prefix="ug-integration-project-") as project, + ): + # Agents walk parent directories for project settings. Keeping cwd out + # of the checkout prevents its .claude/AGENTS.md from influencing a run. + user = UserSession( + Path(temporary), Path(project), installed_binary, root / "artifacts" / case + ) + try: + yield user + finally: + # Restore machine-level settings through the same public CLI that + # created them. A later fresh-home test must not inherit this setup. + if any( + (user.home / ".ucode" / name).is_file() + for name in ("state.json", "managed-backups/manifest.json") + ): + with TerminalProcess( + user, "ug", [str(user.binary), "revert"], "cleanup-revert" + ) as terminal: + terminal.finish() + + +@pytest.fixture +def live_session(session, workspace): + for binary in ["databricks", *os.environ["UG_INTEGRATION_AGENTS"].split(",")]: + if not shutil.which(binary, path=session.env["PATH"]): + pytest.fail(f"Required integration binary is missing: {binary}") + session.env["DATABRICKS_BEARER"] = os.environ["DATABRICKS_BEARER"] + return session + + +@pytest.fixture(scope="session") +def claude_provider(): + return os.environ["UG_INTEGRATION_CLAUDE_PROVIDER"] + + +@pytest.fixture(scope="session") +def codex_provider(): + return os.environ["UG_INTEGRATION_CODEX_PROVIDER"] diff --git a/tests/integration/pytest.ini b/tests/integration/pytest.ini new file mode 100644 index 000000000..ed06ccb80 --- /dev/null +++ b/tests/integration/pytest.ini @@ -0,0 +1,9 @@ +[pytest] +testpaths = . +addopts = --strict-markers --tb=short +markers = + installation: installed-package checks that require no workspace + live: requires the real workspace used by the existing e2e suite + tui: real interactive terminal boot, keyboard input, exit and reopen + claude: only runs when Claude Code is explicitly selected + codex: only runs when Codex is explicitly selected diff --git a/tests/integration/test_installation.py b/tests/integration/test_installation.py new file mode 100644 index 000000000..2f8312d4b --- /dev/null +++ b/tests/integration/test_installation.py @@ -0,0 +1,35 @@ +"""Smoke checks for the installed distribution, outside the source checkout.""" + +import pytest + +pytestmark = pytest.mark.installation + + +def test_ug_installed_wheel_exposes_help_and_version(session): + """Scenario: invoke the freshly installed ug console script. + + Expected: public commands are listed and the installed version is reported. + """ + output = session.run("--help").stdout + assert "configure" in output and "revert" in output + assert session.run("--version").stdout.strip() + + +def test_ug_status_in_fresh_home_is_unconfigured(session): + """Scenario: inspect ug before any setup in a fresh home. + + Expected: status reports Not Configured and does not create saved state. + """ + assert "Not Configured" in session.run("status").stdout + assert not (session.home / ".ucode/state.json").exists() + + +def test_ug_auth_without_configuration_explains_how_to_configure(session): + """Scenario: request an auth token before configuring ug. + + Expected: nonzero exit and configure guidance, without a Python traceback. + """ + result = session.run("auth-token", ok=False) + assert result.returncode != 0 + assert "configure" in result.stdout + result.stderr + assert "Traceback" not in result.stdout + result.stderr diff --git a/tests/integration/test_smart_routing_claude.py b/tests/integration/test_smart_routing_claude.py new file mode 100644 index 000000000..1834760e1 --- /dev/null +++ b/tests/integration/test_smart_routing_claude.py @@ -0,0 +1,106 @@ +"""CUJs: real first-prompt and child-task routing in Claude's TUI.""" + +import pytest +from utils.evidence import FileTask, assert_subagent_routed +from utils.terminal import AgentTerminal + +pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.claude] + + +def test_smart_routing_claude_first_prompt(live_session, workspace): + """Scenario: enable smart routing and submit Claude's first interactive prompt. + + Expected: a real gateway decision selects a model, the prompt is replayed, + and the assistant completes the task before a normal exit. Boot alone fails. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + command = [str(session.binary), "claude", "--enable-smart-routing"] + with AgentTerminal(session, "claude", command, "first-prompt") as tui: + tui.boot() + assert "[ROUTE]" not in session.routing_log("claude"), "Boot must not route a prompt" + tui.submit(task.prompt) + tui.wait_for_task(task) + log = session.routing_log("claude") + assert "[ROUTE] first prompt ->" in log, log + assert "[REPLAY] first prompt submitted" in log, log + tui.exit_normally() + task.assert_completed(session, "claude") + with AgentTerminal(session, "claude", command, "reopen") as tui: + tui.boot() + tui.check_input_and_exit() + + +def test_smart_routing_claude_subagent(live_session, workspace): + """Scenario: ask Claude to delegate a file-reading task with routing enabled. + + Expected: a real child session has a correlated gateway routing decision, + the child returns the file value, and the parent returns that result. + Child model identity remains explicitly unknown if its event omits it. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + command = [str(session.binary), "claude", "--enable-smart-routing"] + with AgentTerminal(session, "claude", command, "subagent-task") as tui: + tui.boot() + tui.submit(task.delegate_prompt) + tui.wait_for_task(task, timeout=240) + task.assert_completed(session, "claude", child=True) + assert_subagent_routed(session, "claude", task) + tui.exit_normally() + task.assert_completed(session, "claude") + + +def test_smart_routing_claude_explicit_model_bypasses_routing(live_session, workspace): + """Scenario: request an explicit model while launching an interactive routing session. + + Expected: claude completes the real task with the caller's model choice, + starts no routing wrapper, exits normally, and reopens with usable input. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + model = session.model_for_explicit_case("claude") + command = [str(session.binary), "claude", "--enable-smart-routing", "--model", model] + with AgentTerminal(session, "claude", command, "explicit-model") as tui: + tui.boot() + tui.submit(task.prompt) + tui.wait_for_task(task) + tui.exit_normally() + task.assert_completed(session, "claude") + session.assert_not_routed() + with AgentTerminal(session, "claude", command, "reopen") as tui: + tui.boot() + tui.check_input_and_exit() + session.assert_not_routed() diff --git a/tests/integration/test_smart_routing_codex.py b/tests/integration/test_smart_routing_codex.py new file mode 100644 index 000000000..64045d8e3 --- /dev/null +++ b/tests/integration/test_smart_routing_codex.py @@ -0,0 +1,106 @@ +"""CUJs: real first-prompt and child-task routing in Codex's TUI.""" + +import pytest +from utils.evidence import FileTask, assert_subagent_routed +from utils.terminal import AgentTerminal + +pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.codex] + + +def test_smart_routing_codex_first_prompt(live_session, workspace): + """Scenario: enable smart routing and submit Codex's first interactive prompt. + + Expected: a real gateway decision selects a model and Codex completes the + file-reading task before a normal exit. A routing fallback does not pass. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + command = [str(session.binary), "codex", "--enable-smart-routing"] + with AgentTerminal(session, "codex", command, "first-prompt") as tui: + tui.boot() + assert "[ROUTE]" not in session.routing_log("codex"), "Boot must not route a prompt" + tui.submit(task.prompt) + tui.wait_for_task(task) + log = session.routing_log("codex") + assert "[ROUTE] selected" in log, log + assert "selection failed" not in log and "settings/update timed out" not in log, log + tui.exit_normally() + task.assert_completed(session, "codex") + with AgentTerminal(session, "codex", command, "reopen") as tui: + tui.boot() + tui.check_input_and_exit() + + +def test_smart_routing_codex_subagent(live_session, workspace): + """Scenario: ask Codex to delegate a file-reading task with routing enabled. + + Expected: a real child session has a correlated gateway routing decision, + the child returns the file value, and the parent returns that result. + The native child turn must use the routed model; inherited parent turns do not count. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + command = [str(session.binary), "codex", "--enable-smart-routing"] + with AgentTerminal(session, "codex", command, "subagent-task") as tui: + tui.boot() + tui.submit(task.delegate_prompt) + tui.wait_for_task(task, timeout=240) + task.assert_completed(session, "codex", child=True) + assert_subagent_routed(session, "codex", task) + tui.exit_normally() + task.assert_completed(session, "codex") + + +def test_smart_routing_codex_explicit_model_bypasses_routing(live_session, workspace): + """Scenario: request an explicit model while launching an interactive routing session. + + Expected: codex completes the real task with the caller's model choice, + starts no routing wrapper, exits normally, and reopens with usable input. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + model = session.model_for_explicit_case("codex") + command = [str(session.binary), "codex", "--enable-smart-routing", "--", "--model", model] + with AgentTerminal(session, "codex", command, "explicit-model") as tui: + tui.boot() + tui.submit(task.prompt) + tui.wait_for_task(task) + tui.exit_normally() + task.assert_completed(session, "codex") + session.assert_not_routed() + with AgentTerminal(session, "codex", command, "reopen") as tui: + tui.boot() + tui.check_input_and_exit() + session.assert_not_routed() diff --git a/tests/integration/test_ug_claude_commands.py b/tests/integration/test_ug_claude_commands.py new file mode 100644 index 000000000..84cfbaf5f --- /dev/null +++ b/tests/integration/test_ug_claude_commands.py @@ -0,0 +1,55 @@ +"""CUJs for inspecting claude commands through ug.""" + +import pytest + +pytestmark = [pytest.mark.live, pytest.mark.claude] + + +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_claude_auth_help(live_session, workspace, routing): + """Scenario: configure claude, then ask ug for auth subcommand help. + + Expected: the real claude help is returned, with no routing wrapper or + model request. This verifies command dispatch, not an interactive session. + """ + session = live_session + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + expected = session.run("auth", "--help", binary="claude").stdout.strip() + actual = session.run("claude", "--", "auth", "--help").stdout + assert expected and expected in actual, actual + session.assert_not_routed() + + +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_claude_mcp_help(live_session, workspace, routing): + """Scenario: configure claude, then ask ug for mcp subcommand help. + + Expected: the real claude help is returned, with no routing wrapper or + model request. This verifies command dispatch, not an interactive session. + """ + session = live_session + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + expected = session.run("mcp", "--help", binary="claude").stdout.strip() + actual = session.run("claude", "--", "mcp", "--help").stdout + assert expected and expected in actual, actual + session.assert_not_routed() diff --git a/tests/integration/test_ug_claude_headless.py b/tests/integration/test_ug_claude_headless.py new file mode 100644 index 000000000..0f537838a --- /dev/null +++ b/tests/integration/test_ug_claude_headless.py @@ -0,0 +1,225 @@ +"""CUJs for using claude from scripts through installed ug.""" + +import json + +import pytest +from utils.evidence import FileTask + +pytestmark = [pytest.mark.live, pytest.mark.claude] + + +def test_ug_claude_headless_prompt_argument(live_session, workspace): + """Scenario: configure claude and submit a headless prompt via argument. + + Expected: the real agent reads the fixture and returns its unknown value in + its structured completed answer, with exit code zero. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + result = session.run( + "claude", + "--", + "-p", + task.prompt, + "--output-format", + "json", + "--allowedTools", + "Read", + timeout=180, + ) + task.assert_headless_answer("claude", result) + session.assert_not_routed() + + +def test_ug_claude_headless_prompt_stdin(live_session, workspace): + """Scenario: configure claude and submit a headless prompt via stdin. + + Expected: the real agent reads the fixture and returns its unknown value in + its structured completed answer, with exit code zero. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + result = session.run( + "claude", + "--", + "-p", + "--output-format", + "json", + "--allowedTools", + "Read", + timeout=180, + input_text=task.prompt + "\n", + ) + task.assert_headless_answer("claude", result) + session.assert_not_routed() + + +def test_ug_claude_headless_prompt_after_separator(live_session, workspace): + """Scenario: configure claude and submit a headless prompt via after separator. + + Expected: the real agent reads the fixture and returns its unknown value in + its structured completed answer, with exit code zero. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + result = session.run( + "claude", + "--", + "-p", + "--output-format", + "json", + "--allowedTools", + "Read", + "--", + task.prompt, + timeout=180, + ) + task.assert_headless_answer("claude", result) + session.assert_not_routed() + + +@pytest.mark.parametrize("model_form", ["separate", "equals"]) +def test_ug_claude_headless_explicit_model_bypasses_routing(live_session, workspace, model_form): + """Scenario: choose an explicit model while global smart routing is enabled. + + Expected: the model option is accepted, the real file task completes, and + no routing wrapper overrides the caller's choice. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + model = session.model_for_explicit_case("claude") + model_args = ["--model", model] if model_form == "separate" else [f"--model={model}"] + session.env["ENABLE_SMART_ROUTING_V2"] = "1" + result = session.run( + "claude", + "--", + "-p", + task.prompt, + "--output-format", + "json", + "--allowedTools", + "Read", + *model_args, + timeout=180, + ) + task.assert_headless_answer("claude", result) + session.assert_not_routed() + + +def test_ug_claude_preserves_caller_settings_and_hook(live_session, workspace): + """Scenario: a launcher passes a settings path containing spaces to ug claude. + + Expected: the caller's real SessionStart hook executes, its input file stays + unchanged, and gateway authentication still supports a completed file task. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + settings = session.cwd / "caller settings.json" + # Ordinary user-owned input, not fabricated ug state or generated gateway config. + content = json.dumps( + { + "hooks": { + "SessionStart": [ + { + "hooks": [ + {"type": "command", "command": "echo caller-hook-ran > caller-hook.txt"} + ] + } + ] + } + } + ) + settings.write_text(content) + result = session.run( + "claude", + "--", + "-p", + task.prompt, + "--output-format", + "json", + "--allowedTools", + "Read", + "--settings", + str(settings), + timeout=180, + ) + task.assert_headless_answer("claude", result) + assert (session.cwd / "caller-hook.txt").read_text().strip() == "caller-hook-ran" + assert settings.read_text() == content + + +def test_ug_claude_reports_unsupported_short_model_option(live_session, workspace): + """Scenario: pass -m to the selected Claude version, which does not support it. + + Expected: ug preserves the actual agent's unknown-option error and status. + This is an error-reporting journey, not a successful inference claim. + """ + session = live_session + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + expected = session.run("-m", "sonnet", "-p", "hi", binary="claude", ok=False) + actual = session.run("claude", "--", "-m", "sonnet", "-p", "hi", ok=False) + assert expected.returncode != 0 and "unknown option '-m'" in expected.stderr + assert actual.returncode == expected.returncode + assert "unknown option '-m'" in actual.stderr diff --git a/tests/integration/test_ug_codex_app_server.py b/tests/integration/test_ug_codex_app_server.py new file mode 100644 index 000000000..9a9c3559b --- /dev/null +++ b/tests/integration/test_ug_codex_app_server.py @@ -0,0 +1,34 @@ +"""CUJ: a client connects to Codex's real app-server through ug.""" + +import pytest + +pytestmark = [pytest.mark.live, pytest.mark.codex] + + +@pytest.mark.parametrize("separator", [False, True], ids=["direct", "launcher-separator"]) +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_app_server_client_initializes(live_session, workspace, separator, routing): + """Scenario: configure Codex and connect a real stdio client to ug codex app-server. + + Expected: initialize returns a valid JSON-RPC result, diagnostics stay off + the protocol stream, and this utility command never starts smart routing. + """ + session = live_session + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + + args = ["app-server", "--listen", "stdio://"] + if separator: + args.insert(0, "--") + response = session.app_server_handshake(args) + assert response["result"]["userAgent"] + session.assert_not_routed() diff --git a/tests/integration/test_ug_codex_commands.py b/tests/integration/test_ug_codex_commands.py new file mode 100644 index 000000000..2e6a32de7 --- /dev/null +++ b/tests/integration/test_ug_codex_commands.py @@ -0,0 +1,134 @@ +"""CUJs for inspecting codex commands through ug.""" + +import pytest + +pytestmark = [pytest.mark.live, pytest.mark.codex] + + +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_app_help(live_session, workspace, routing): + """Scenario: configure codex, then ask ug for app subcommand help. + + Expected: the real codex help is returned, with no routing wrapper or + model request. This verifies command dispatch, not an interactive session. + """ + session = live_session + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + expected = session.run("app", "--help", binary="codex").stdout.strip() + actual = session.run("codex", "--", "app", "--help").stdout + assert expected and expected in actual, actual + session.assert_not_routed() + + +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_app_server_help(live_session, workspace, routing): + """Scenario: configure codex, then ask ug for app-server subcommand help. + + Expected: the real codex help is returned, with no routing wrapper or + model request. This verifies command dispatch, not an interactive session. + """ + session = live_session + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + expected = session.run("app-server", "--help", binary="codex").stdout.strip() + actual = session.run("codex", "--", "app-server", "--help").stdout + assert expected and expected in actual, actual + session.assert_not_routed() + + +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_exec_help(live_session, workspace, routing): + """Scenario: configure codex, then ask ug for exec subcommand help. + + Expected: the real codex help is returned, with no routing wrapper or + model request. This verifies command dispatch, not an interactive session. + """ + session = live_session + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + expected = session.run("exec", "--help", binary="codex").stdout.strip() + actual = session.run("codex", "--", "exec", "--help").stdout + assert expected and expected in actual, actual + session.assert_not_routed() + + +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_mcp_help(live_session, workspace, routing): + """Scenario: configure codex, then ask ug for mcp subcommand help. + + Expected: the real codex help is returned, with no routing wrapper or + model request. This verifies command dispatch, not an interactive session. + """ + session = live_session + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + expected = session.run("mcp", "--help", binary="codex").stdout.strip() + actual = session.run("codex", "--", "mcp", "--help").stdout + assert expected and expected in actual, actual + session.assert_not_routed() + + +@pytest.mark.parametrize("routing", ["0", "1"], ids=["routing-off", "routing-on"]) +def test_ug_codex_app_reports_unknown_argument(live_session, workspace, routing): + """Scenario: pass an unknown option directly to ug codex app. + + Expected: the actual Codex parser's error and exit status are preserved, + without opening a desktop application or starting routing. + """ + session = live_session + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + session.env["ENABLE_SMART_ROUTING_V2"] = routing + args = ["app", "--ug-integration-unknown-option"] + expected = session.run(*args, binary="codex", ok=False) + actual = session.run("codex", *args, ok=False) + assert expected.returncode != 0 and "error:" in expected.stderr + assert actual.returncode == expected.returncode + parser_error = expected.stderr[expected.stderr.index("error:") :].strip() + assert parser_error in actual.stderr, actual.stdout + actual.stderr + session.assert_not_routed() diff --git a/tests/integration/test_ug_codex_headless.py b/tests/integration/test_ug_codex_headless.py new file mode 100644 index 000000000..0b83f54d1 --- /dev/null +++ b/tests/integration/test_ug_codex_headless.py @@ -0,0 +1,129 @@ +"""CUJs for using codex from scripts through installed ug.""" + +import pytest +from utils.evidence import FileTask + +pytestmark = [pytest.mark.live, pytest.mark.codex] + + +def test_ug_codex_headless_prompt_argument(live_session, workspace): + """Scenario: configure codex and submit a headless prompt via argument. + + Expected: the real agent reads the fixture and returns its unknown value in + its structured completed answer, with exit code zero. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + result = session.run( + "codex", "--", "exec", "--skip-git-repo-check", "--json", task.prompt, timeout=180 + ) + task.assert_headless_answer("codex", result) + session.assert_not_routed() + + +def test_ug_codex_headless_prompt_stdin(live_session, workspace): + """Scenario: configure codex and submit a headless prompt via stdin. + + Expected: the real agent reads the fixture and returns its unknown value in + its structured completed answer, with exit code zero. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + result = session.run( + "codex", + "--", + "exec", + "--skip-git-repo-check", + "--json", + "-", + timeout=180, + input_text=task.prompt + "\n", + ) + task.assert_headless_answer("codex", result) + session.assert_not_routed() + + +def test_ug_codex_headless_prompt_after_separator(live_session, workspace): + """Scenario: configure codex and submit a headless prompt via after separator. + + Expected: the real agent reads the fixture and returns its unknown value in + its structured completed answer, with exit code zero. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + + result = session.run( + "codex", "--", "exec", "--skip-git-repo-check", "--json", "--", task.prompt, timeout=180 + ) + task.assert_headless_answer("codex", result) + session.assert_not_routed() + + +@pytest.mark.parametrize("model_form", ["separate", "equals", "short"]) +def test_ug_codex_headless_explicit_model_bypasses_routing(live_session, workspace, model_form): + """Scenario: choose an explicit model while global smart routing is enabled. + + Expected: the model option is accepted, the real file task completes, and + no routing wrapper overrides the caller's choice. + """ + session = live_session + task = FileTask(session) + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + model = session.model_for_explicit_case("codex") + model_args = ["--model", model] if model_form == "separate" else [f"--model={model}"] + if model_form == "short": + model_args = ["-m", model] + session.env["ENABLE_SMART_ROUTING_V2"] = "1" + result = session.run( + "codex", + "--", + "exec", + "--skip-git-repo-check", + "--json", + task.prompt, + *model_args, + timeout=180, + ) + task.assert_headless_answer("codex", result) + session.assert_not_routed() diff --git a/tests/integration/test_ug_configure_claude.py b/tests/integration/test_ug_configure_claude.py new file mode 100644 index 000000000..37fc5ac5c --- /dev/null +++ b/tests/integration/test_ug_configure_claude.py @@ -0,0 +1,82 @@ +"""CUJs: configure Claude through ug, then use its real interactive session.""" + +import pytest +from utils.evidence import FileTask +from utils.terminal import AgentTerminal, ConfigureTerminal + +pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.claude] + + +def test_ug_configure_claude_databricks(live_session, workspace): + """Scenario: configure Claude with Databricks Hosted and use its TUI. + + Expected: configure succeeds, Claude returns a file value through the real + gateway, exits normally, and can reopen the configuration ug created. + Optional AI Tools are disabled; the selected agent version is kept pinned. + """ + session = live_session + task = FileTask(session) + + # Configure using the installed public CLI, including its normal validation. + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-upgrade", + "--disable-databricks-ai-tools", + timeout=240, + ) + assert not session.workspace_state().get("provider_services", {}).get("claude") + + # Use the real TUI; a config file or startup banner alone is not success. + with AgentTerminal(session, "claude", [str(session.binary), "claude"], "first-session") as tui: + tui.boot() + tui.submit(task.prompt) + tui.wait_for_task(task) + tui.exit_normally() + task.assert_completed(session, "claude") + session.assert_not_routed() + + # Reopen the same home, without configuring again or seeding onboarding state. + with AgentTerminal(session, "claude", [str(session.binary), "claude"], "reopen") as tui: + tui.boot() + tui.check_input_and_exit() + + +def test_ug_configure_claude_anthropic_mps(live_session, workspace, claude_provider): + """Scenario: choose the real Anthropic MPS in ug configure's provider picker. + + Expected: ug saves that provider, and launching Claude without --provider + uses the saved choice to complete a file-reading task and exit normally. + """ + session = live_session + task = FileTask(session) + + # Provider selection is interactive; --agents would bypass this real picker. + command = [ + str(session.binary), + "configure", + "--workspaces", + workspace, + "--skip-upgrade", + "--disable-databricks-ai-tools", + ] + with ConfigureTerminal(session, "claude", command, "configure-provider") as configure: + configure.select_agent("Claude Code") + configure.choose("How should Claude Code get its models?", "External Models") + configure.choose("Select a model provider service:", claude_provider) + configure.finish(timeout=240) + assert session.workspace_state()["provider_services"]["claude"] == claude_provider + assert claude_provider in session.run("status").stdout + + with AgentTerminal( + session, "claude", [str(session.binary), "claude"], "provider-session" + ) as tui: + tui.boot() + tui.submit(task.prompt) + tui.wait_for_task(task, timeout=300) + tui.exit_normally() + task.assert_completed(session, "claude") + session.assert_not_routed() diff --git a/tests/integration/test_ug_configure_claude_lifecycle.py b/tests/integration/test_ug_configure_claude_lifecycle.py new file mode 100644 index 000000000..666e95a59 --- /dev/null +++ b/tests/integration/test_ug_configure_claude_lifecycle.py @@ -0,0 +1,106 @@ +"""CUJs for changing and undoing claude's ug setup.""" + +import json + +import pytest +from utils.evidence import FileTask +from utils.terminal import TerminalProcess + +pytestmark = [pytest.mark.live, pytest.mark.claude] + + +def test_ug_configure_claude_repeat_and_revert(live_session, workspace): + """Scenario: configure twice over user-owned settings, use the agent, then revert. + + Expected: repeat setup remains usable, unrelated settings survive, no bearer + is saved in ug state, and repeated revert restores the original configuration. + The generated ug file must be removed; a leftover file remains a failure. + """ + session = live_session + task = FileTask(session) + # Ordinary pre-existing user settings, not manufactured gateway state. + user_path = session.home / ".claude/settings.json" + original = '{"permissions":{"allow":["Read"]},"env":{"UG_USER_SETTING":"keep"}}\n' + user_path.parent.mkdir(parents=True) + user_path.write_text(original) + + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + first = session.workspace_state() + assert "claude" in first["available_tools"] + assert first["claude_models"] + session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + assert session.workspace_state()["available_tools"] == first["available_tools"] + assert workspace in session.run("status").stdout + current = json.loads(user_path.read_text()) + assert all(current.get(key) == value for key, value in json.loads(original).items()) + bearer_was_saved = any( + session.env["DATABRICKS_BEARER"] in path.read_text(errors="replace") + for path in (session.home / ".ucode").rglob("*") + if path.is_file() + ) + assert not bearer_was_saved, "The workspace bearer was saved in ug state" + + result = session.run( + "claude", + "--", + "-p", + task.prompt, + "--output-format", + "json", + "--allowedTools", + "Read", + timeout=180, + ) + task.assert_headless_answer("claude", result) + with TerminalProcess(session, "ug", [str(session.binary), "revert"], "revert") as terminal: + terminal.finish() + assert json.loads(user_path.read_text()) == json.loads(original) + assert "Not Configured" in session.run("status").stdout + assert not (session.home / ".claude/ucode-settings.json").exists() + with TerminalProcess( + session, "ug", [str(session.binary), "revert"], "repeat-revert" + ) as terminal: + terminal.finish() + + +def test_ug_configure_claude_rejects_invalid_credentials(live_session, workspace): + """Scenario: configure claude against the real workspace with a rejected bearer. + + Expected: authentication fails clearly and no successful setup is saved. + """ + session = live_session + session.env["DATABRICKS_BEARER"] = "ug-integration-intentionally-invalid" + result = session.run( + "configure", + "--agents", + "claude", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ok=False, + ) + assert result.returncode != 0 + output = result.stdout + result.stderr + assert "rejected the access token" in output or "401" in output, output + assert "Configuration Complete" not in output + assert not (session.home / ".ucode/state.json").exists() diff --git a/tests/integration/test_ug_configure_codex.py b/tests/integration/test_ug_configure_codex.py new file mode 100644 index 000000000..c2759c604 --- /dev/null +++ b/tests/integration/test_ug_configure_codex.py @@ -0,0 +1,76 @@ +"""CUJs: configure Codex through ug, then use its real interactive session.""" + +import pytest +from utils.evidence import FileTask +from utils.terminal import AgentTerminal, ConfigureTerminal + +pytestmark = [pytest.mark.live, pytest.mark.tui, pytest.mark.codex] + + +def test_ug_configure_codex_databricks(live_session, workspace): + """Scenario: configure Codex with Databricks Hosted and use its TUI. + + Expected: configure succeeds, Codex returns a file value through the real + gateway, exits normally, and can reopen the configuration ug created. + Optional AI Tools are disabled; the selected agent version is kept pinned. + """ + session = live_session + task = FileTask(session) + + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-upgrade", + "--disable-databricks-ai-tools", + timeout=240, + ) + assert not session.workspace_state().get("provider_services", {}).get("codex") + + with AgentTerminal(session, "codex", [str(session.binary), "codex"], "first-session") as tui: + tui.boot() + tui.submit(task.prompt) + tui.wait_for_task(task) + tui.exit_normally() + task.assert_completed(session, "codex") + session.assert_not_routed() + + with AgentTerminal(session, "codex", [str(session.binary), "codex"], "reopen") as tui: + tui.boot() + tui.check_input_and_exit() + + +def test_ug_configure_codex_openai_mps(live_session, workspace, codex_provider): + """Scenario: choose the real OpenAI MPS in ug configure's provider picker. + + Expected: ug saves that provider, and launching Codex without --provider + uses the saved choice to complete a file-reading task and exit normally. + """ + session = live_session + task = FileTask(session) + + command = [ + str(session.binary), + "configure", + "--workspaces", + workspace, + "--skip-upgrade", + "--disable-databricks-ai-tools", + ] + with ConfigureTerminal(session, "codex", command, "configure-provider") as configure: + configure.select_agent("Codex") + configure.choose("How should Codex get its models?", "External Models") + configure.choose("Select a model provider service:", codex_provider) + configure.finish(timeout=240) + assert session.workspace_state()["provider_services"]["codex"] == codex_provider + assert codex_provider in session.run("status").stdout + + with AgentTerminal(session, "codex", [str(session.binary), "codex"], "provider-session") as tui: + tui.boot() + tui.submit(task.prompt) + tui.wait_for_task(task, timeout=300) + tui.exit_normally() + task.assert_completed(session, "codex") + session.assert_not_routed() diff --git a/tests/integration/test_ug_configure_codex_lifecycle.py b/tests/integration/test_ug_configure_codex_lifecycle.py new file mode 100644 index 000000000..49bb3351b --- /dev/null +++ b/tests/integration/test_ug_configure_codex_lifecycle.py @@ -0,0 +1,98 @@ +"""CUJs for changing and undoing codex's ug setup.""" + +import tomllib + +import pytest +from utils.evidence import FileTask +from utils.terminal import TerminalProcess + +pytestmark = [pytest.mark.live, pytest.mark.codex] + + +def test_ug_configure_codex_repeat_and_revert(live_session, workspace): + """Scenario: configure twice over user-owned settings, use the agent, then revert. + + Expected: repeat setup remains usable, unrelated settings survive, no bearer + is saved in ug state, and repeated revert restores the original configuration. + The generated ug file must be removed; a leftover file remains a failure. + """ + session = live_session + task = FileTask(session) + # Ordinary pre-existing user settings, not manufactured gateway state. + user_path = session.home / ".codex/config.toml" + original = "# user comment\n[notice]\nhide_rate_limit_model_nudge = true\n" + user_path.parent.mkdir(parents=True) + user_path.write_text(original) + + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + first = session.workspace_state() + assert "codex" in first["available_tools"] + assert first["codex_models"] + session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ) + assert session.workspace_state()["available_tools"] == first["available_tools"] + assert workspace in session.run("status").stdout + current = tomllib.loads(user_path.read_text()) + assert all(current.get(key) == value for key, value in tomllib.loads(original).items()) + bearer_was_saved = any( + session.env["DATABRICKS_BEARER"] in path.read_text(errors="replace") + for path in (session.home / ".ucode").rglob("*") + if path.is_file() + ) + assert not bearer_was_saved, "The workspace bearer was saved in ug state" + + result = session.run( + "codex", "--", "exec", "--skip-git-repo-check", "--json", task.prompt, timeout=180 + ) + task.assert_headless_answer("codex", result) + with TerminalProcess(session, "ug", [str(session.binary), "revert"], "revert") as terminal: + terminal.finish() + assert tomllib.loads(user_path.read_text()) == tomllib.loads(original) + assert "Not Configured" in session.run("status").stdout + assert not (session.home / ".codex/ucode.config.toml").exists() + with TerminalProcess( + session, "ug", [str(session.binary), "revert"], "repeat-revert" + ) as terminal: + terminal.finish() + + +def test_ug_configure_codex_rejects_invalid_credentials(live_session, workspace): + """Scenario: configure codex against the real workspace with a rejected bearer. + + Expected: authentication fails clearly and no successful setup is saved. + """ + session = live_session + session.env["DATABRICKS_BEARER"] = "ug-integration-intentionally-invalid" + result = session.run( + "configure", + "--agents", + "codex", + "--workspaces", + workspace, + "--skip-validate", + "--skip-upgrade", + "--disable-databricks-ai-tools", + ok=False, + ) + assert result.returncode != 0 + output = result.stdout + result.stderr + assert "rejected the access token" in output or "401" in output, output + assert "Configuration Complete" not in output + assert not (session.home / ".ucode/state.json").exists() diff --git a/tests/integration/utils/Dockerfile b/tests/integration/utils/Dockerfile new file mode 100644 index 000000000..537818223 --- /dev/null +++ b/tests/integration/utils/Dockerfile @@ -0,0 +1,26 @@ +# Build using the documented minimal archive. Only the runner and tests enter the image; +# pass a release version or mount a prebuilt wheel at runtime. +FROM ghcr.io/astral-sh/uv:0.9.8 AS uv +FROM node:22.19.0-bookworm-slim AS node +FROM python:3.12.10-slim-bookworm + +ARG DATABRICKS_VERSION=1.9.0 +ARG TARGETARCH +RUN apt-get update && apt-get install -y --no-install-recommends ca-certificates curl unzip git libstdc++6 \ + && curl -fL "https://github.com/databricks/cli/releases/download/v${DATABRICKS_VERSION}/databricks_cli_${DATABRICKS_VERSION}_linux_${TARGETARCH:-$(dpkg --print-architecture)}.zip" -o /tmp/databricks.zip \ + && unzip /tmp/databricks.zip -d /tmp/databricks \ + && install /tmp/databricks/databricks /usr/local/bin/databricks \ + && rm -rf /tmp/databricks* /var/lib/apt/lists/* +COPY --from=uv /uv /usr/local/bin/uv +COPY --from=node /usr/local/bin/node /usr/local/bin/node +COPY --from=node /usr/local/lib/node_modules /usr/local/lib/node_modules +RUN ln -s /usr/local/lib/node_modules/npm/bin/npm-cli.js /usr/local/bin/npm \ + && useradd --create-home --uid 10001 integration + +WORKDIR /opt/ug-integration +COPY scripts/run_integration.py scripts/run_integration.py +COPY tests/integration tests/integration +RUN mkdir /results && chown integration:integration /results +USER integration +ENV HOME=/home/integration +ENTRYPOINT ["python", "scripts/run_integration.py", "--output", "/results/run"] diff --git a/tests/integration/utils/Dockerfile.dockerignore b/tests/integration/utils/Dockerfile.dockerignore new file mode 100644 index 000000000..044382fb9 --- /dev/null +++ b/tests/integration/utils/Dockerfile.dockerignore @@ -0,0 +1,8 @@ +** +!scripts/ +!scripts/run_integration.py +!tests/ +!tests/integration/ +!tests/integration/** +tests/integration/__pycache__/ +tests/integration/.pytest_cache/ diff --git a/tests/integration/utils/__init__.py b/tests/integration/utils/__init__.py new file mode 100644 index 000000000..d452b37b6 --- /dev/null +++ b/tests/integration/utils/__init__.py @@ -0,0 +1 @@ +"""Process, terminal, and evidence utilities for the integration CUJs.""" diff --git a/tests/integration/utils/evidence.py b/tests/integration/utils/evidence.py new file mode 100644 index 000000000..6cfafdac7 --- /dev/null +++ b/tests/integration/utils/evidence.py @@ -0,0 +1,197 @@ +"""Read real agent transcripts and ug routing records without changing them.""" + +import json +import uuid +from pathlib import Path + + +def read_jsonl(path: Path) -> list[dict]: + if not path.is_file(): + return [] + text = path.read_text() + lines = text.splitlines(keepends=True) + records = [] + for index, line in enumerate(lines): + try: + value = json.loads(line) + except json.JSONDecodeError: + # A running agent may not have finished its last write yet. + if index == len(lines) - 1 and not line.endswith("\n"): + break + raise + if isinstance(value, dict): + records.append(value) + return records + + +def agent_sessions(session, agent: str) -> dict[str, list[dict]]: + directory = session.home / (".claude/projects" if agent == "claude" else ".codex/sessions") + return { + str(path.relative_to(directory)): read_jsonl(path) for path in directory.rglob("*.jsonl") + } + + +def assistant_answers(agent: str, records: list[dict]) -> list[str]: + answers = [] + for record in records: + if agent == "claude" and record.get("type") == "assistant": + message = record.get("message", {}) + if message.get("role") == "assistant": + answers.extend( + part["text"] + for part in message.get("content", []) + if part.get("type") == "text" and isinstance(part.get("text"), str) + ) + if agent == "codex" and record.get("type") == "event_msg": + payload = record.get("payload", {}) + if payload.get("type") == "task_complete" and payload.get("last_agent_message"): + answers.append(payload["last_agent_message"]) + return answers + + +def is_child_session(agent: str, path: str, records: list[dict]) -> bool: + if agent == "claude": + return "/subagents/" in path + return any( + record.get("type") == "session_meta" + and isinstance(record.get("payload", {}).get("source"), dict) + and "subagent" in record["payload"]["source"] + for record in records + ) + + +class FileTask: + """Ordinary project input; the expected answer is never included in the prompt.""" + + def __init__(self, session): + self.value = uuid.uuid4().hex + self.filename = "input-" + uuid.uuid4().hex[:8] + ".txt" + (session.cwd / self.filename).write_text(self.value + "\n") + self.prompt = f"Read {self.filename} using a tool. Reply with only its contents." + self.delegate_prompt = ( + f"Delegate this task to one subagent: read {self.filename} using a tool and return " + "its contents. Do not read the file yourself. Wait for the subagent and reply " + "with only the value it returned." + ) + + def completed(self, session, agent: str, *, child: bool = False) -> bool: + for path, records in agent_sessions(session, agent).items(): + if is_child_session(agent, path, records) != child: + continue + if any(self.value in text for text in assistant_answers(agent, records)): + return True + return False + + def assert_completed(self, session, agent: str, *, child: bool = False) -> None: + sessions = agent_sessions(session, agent) + session.record("agent-sessions.json", sessions) + assert self.completed(session, agent, child=child), ( + f"No {'child' if child else 'parent'} assistant answer contained the file's value; " + "echoed prompts and tool results do not count as completed answers." + ) + + def assert_headless_answer(self, agent: str, result) -> None: + """Read the real CLI's structured final answer, never its echoed input.""" + payloads = [] + for line in result.stdout.splitlines(): + try: + value = json.loads(line) + except json.JSONDecodeError: + continue # ug may print human-readable launch status before agent JSON. + if isinstance(value, dict): + payloads.append(value) + if agent == "claude": + final = [row for row in payloads if row.get("type") == "result"] + assert final and not final[-1].get("is_error"), result.stdout + assert self.value in final[-1].get("result", ""), result.stdout + else: + assert any(row.get("type") == "turn.completed" for row in payloads), result.stdout + answers = [ + row.get("item", {}).get("text", "") + for row in payloads + if row.get("type") == "item.completed" + and row.get("item", {}).get("type") == "agent_message" + ] + assert any(self.value in answer for answer in answers), result.stdout + + +def assert_subagent_routed(session, agent: str, task: FileTask) -> None: + """Require a real gateway decision correlated with an actual child start.""" + root = session.home / ".ucode" + decisions = read_jsonl(root / f"{agent}-smart-routing-decisions.jsonl") + audit = read_jsonl(root / f"{agent}-smart-routing-audit.jsonl") + session.record("subagent-routing.json", {"decisions": decisions, "starts": audit}) + assert decisions, "No real subagent routing decision was recorded" + for decision in decisions: + assert decision.get("requested_model") and decision.get("router_model"), decision + if agent == "codex": + # Codex exposes parent linkage and the child's actual turn model in its + # native rollouts. Its ug SubagentStart audit can be empty even when the + # child ran. Match native evidence, including the completed file task. + sessions = agent_sessions(session, agent) + linked = [] + for path, records in sessions.items(): + metadata = next( + (row["payload"] for row in records if row.get("type") == "session_meta"), {} + ) + source = metadata.get("source") + if not isinstance(source, dict): + continue + parent_id = source.get("subagent", {}).get("thread_spawn", {}).get("parent_thread_id") + if not parent_id: + continue + # A child rollout starts with inherited parent history. Exclude + # those turn IDs so a parent's answer/model cannot satisfy this check. + parent_turn_ids = set() + for other in sessions.values(): + first_meta = next( + (row["payload"] for row in other if row.get("type") == "session_meta"), {} + ) + if first_meta.get("id") == parent_id: + parent_turn_ids.update( + row["payload"]["turn_id"] + for row in other + if row.get("type") == "turn_context" and row["payload"].get("turn_id") + ) + assert parent_turn_ids, f"No native parent turns found for {parent_id}" + for decision in decisions: + if decision.get("session_id") != parent_id: + continue + routed_turn_ids = { + row["payload"]["turn_id"] + for row in records + if row.get("type") == "turn_context" + and row["payload"].get("model") == decision["requested_model"] + and row["payload"].get("turn_id") + } - parent_turn_ids + for row in records: + payload = row.get("payload", {}) + if ( + row.get("type") == "event_msg" + and payload.get("type") == "task_complete" + and payload.get("turn_id") in routed_turn_ids + and task.value in (payload.get("last_agent_message") or "") + ): + linked.append( + { + "decision_id": decision["decision_id"], + "parent_id": parent_id, + "child_id": metadata["id"], + "path": path, + "turn_id": payload["turn_id"], + "model": decision["requested_model"], + } + ) + session.record("subagent-routing.json", {"decisions": decisions, "native_children": linked}) + assert linked, "No routed native child turn completed the delegated file task" + assert {row["decision_id"] for row in linked} == { + row["decision_id"] for row in decisions + }, "A routing decision had no matching completed child turn" + return + decision_ids = {decision["decision_id"] for decision in decisions} + routed_starts = [row for row in audit if row.get("decision_id") in decision_ids] + assert routed_starts and all(row.get("agent_id") for row in routed_starts), audit + assert all(row.get("matches_router_decision") is not False for row in routed_starts), audit + # Some agent versions omit the child's model from SubagentStart. The report + # preserves that unknown value; this test claims decision + spawn + task, + # not model-identity verification when the agent did not expose it. diff --git a/tests/integration/utils/harness.py b/tests/integration/utils/harness.py new file mode 100644 index 000000000..cbf9c3590 --- /dev/null +++ b/tests/integration/utils/harness.py @@ -0,0 +1,261 @@ +"""Drive installed programs; never import the application under test.""" + +from __future__ import annotations + +import contextlib +import json +import os +import queue +import re +import signal +import subprocess +import threading +import time +from pathlib import Path + +ANSI = re.compile(r"\x1b\[[0-9;?]*[ -/]*[@-~]") + + +def clean_environment(home: Path) -> dict[str, str]: + # An allowlist prevents a developer's agent keys, settings, plugins, Python + # imports, and routing flags from silently changing the tested combination. + keep = ( + "PATH", + "SYSTEMROOT", + "COMSPEC", + "LANG", + "LC_ALL", + "SSL_CERT_FILE", + "SSL_CERT_DIR", + "REQUESTS_CA_BUNDLE", + "NODE_EXTRA_CA_CERTS", + "HTTPS_PROXY", + "HTTP_PROXY", + "NO_PROXY", + ) + env = {key: os.environ[key] for key in keep if key in os.environ} + env.update( + { + "HOME": str(home), + "USERPROFILE": str(home), + "XDG_CONFIG_HOME": str(home / ".config"), + "XDG_CACHE_HOME": str(home / ".cache"), + "XDG_DATA_HOME": str(home / ".local/share"), + "CLAUDE_CONFIG_DIR": str(home / ".claude"), + "CODEX_HOME": str(home / ".codex"), + "DATABRICKS_CONFIG_FILE": str(home / ".databrickscfg"), + "DISABLE_AUTOUPDATER": "1", + "CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1", + "NO_COLOR": "1", + "TERM": "dumb", + "COLUMNS": "160", + "ENABLE_SMART_ROUTING_V2": "0", + "PYTHONNOUSERSITE": "1", + } + ) + return env + + +def stop_process(proc: subprocess.Popen) -> None: + """Reap the entire process group, including servers left by a failed agent.""" + if os.name == "posix": + with contextlib.suppress(ProcessLookupError): + os.killpg(proc.pid, signal.SIGTERM) + try: + proc.wait(timeout=5) + except subprocess.TimeoutExpired: + pass + # The leader may have exited while a grandchild kept running. + with contextlib.suppress(ProcessLookupError): + os.killpg(proc.pid, signal.SIGKILL) + elif proc.poll() is None: + proc.kill() + proc.wait(timeout=5) + + +class UserSession: + def __init__(self, root: Path, project_root: Path, binary: Path, artifacts: Path): + self.home = root / "home" + self.cwd = project_root / "project with spaces" + self.home.mkdir(parents=True) + self.cwd.mkdir() + self.binary = binary + self.env = clean_environment(self.home) + self.artifacts = artifacts + self.artifacts.mkdir(parents=True, exist_ok=True) + self.commands = 0 + + def redact(self, text: str) -> str: + for token in (os.environ.get("DATABRICKS_BEARER"), self.env.get("DATABRICKS_BEARER")): + if token: + text = text.replace(token, "") + return ANSI.sub("", text) + + def run( + self, + *args: str, + timeout: int = 120, + ok: bool = True, + binary=None, + input_text: str | None = None, + ): + command = [str(binary or self.binary), *args] + proc = subprocess.Popen( + command, + cwd=self.cwd, + env=self.env, + stdin=subprocess.PIPE if input_text is not None else subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + start_new_session=os.name == "posix", + ) + timed_out = False + try: + stdout, stderr = proc.communicate(input=input_text, timeout=timeout) + except subprocess.TimeoutExpired: + timed_out = True + stop_process(proc) + stdout, stderr = proc.communicate(timeout=5) + finally: + stop_process(proc) + result = subprocess.CompletedProcess( + command, proc.returncode, self.redact(stdout), self.redact(stderr) + ) + self.commands += 1 + self.record( + f"command-{self.commands}.json", + { + "argv": command, + "returncode": result.returncode, + "timed_out": timed_out, + "stdin": input_text, + "stdout": result.stdout, + "stderr": result.stderr, + }, + ) + detail = f"{command}\n{result.stdout}\n{result.stderr}" + assert not timed_out, f"Command exceeded {timeout}s:\n{detail}" + if ok: + assert result.returncode == 0, detail + return result + + def record(self, name: str, value: object) -> None: + (self.artifacts / name).write_text(self.redact(json.dumps(value, indent=2))) + + def state(self) -> dict: + return json.loads((self.home / ".ucode/state.json").read_text()) + + def workspace_state(self) -> dict: + state = self.state() + return state["workspaces"][state["current_workspace"]] + + def routing_log(self, agent: str) -> str: + name = "claude-v2-pty.log" if agent == "claude" else "codex-v2-interposer.log" + path = self.home / ".ucode" / name + assert path.is_file(), f"No routing log was written: {name}" + value = path.read_text() + self.record(name + ".json", {"log": value}) + return value + + def model_for_explicit_case(self, agent: str) -> str: + """Use a real discovered model only when testing an explicit model option.""" + model = os.environ.get(f"UG_INTEGRATION_{agent.upper()}_MODEL", "").strip() + source = "runner override" + if not model: + state = self.state() + workspace = state["workspaces"][state["current_workspace"]] + models = workspace[f"{agent}_models"] + values = models.values() if isinstance(models, dict) else models + prefix = "system.ai.claude-" if agent == "claude" else "system.ai.gpt-" + model = next((value for value in values if value.startswith(prefix)), "") + source = "ug configure discovery" + assert model, ( + f"ug configure found no system.ai model for {agent}; use --{agent}-model to reproduce a specific model." + ) + self.record("model.json", {"model": model, "source": source}) + return model + + def assert_not_routed(self) -> None: + # These are user-visible diagnostics produced only by routing wrappers. + for name in ("codex-v2-interposer.log", "claude-v2-pty.log"): + assert not (self.home / ".ucode" / name).exists(), f"Unexpected routing: {name}" + + def app_server_handshake(self, args: list[str], timeout: int = 120) -> dict: + """Speak the real Codex stdio protocol and require an initialize response.""" + command = [str(self.binary), "codex", *args] + messages: queue.Queue = queue.Queue() + transcript: list[str] = [] + diagnostics: list[str] = [] + proc = subprocess.Popen( + command, + cwd=self.cwd, + env=self.env, + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + bufsize=1, + start_new_session=os.name == "posix", + ) + + def read_output(): + for line in proc.stdout: + transcript.append(self.redact(line)) + try: + messages.put(json.loads(line)) + except ValueError: + if line.strip(): + messages.put({"protocol_error": self.redact(line)}) + messages.put(None) + + def read_diagnostics(): + for line in proc.stderr: + diagnostics.append(self.redact(line)) + + reader = threading.Thread(target=read_output, daemon=True) + stderr_reader = threading.Thread(target=read_diagnostics, daemon=True) + reader.start() + stderr_reader.start() + try: + proc.stdin.write( + json.dumps( + { + "id": 1, + "method": "initialize", + "params": {"clientInfo": {"name": "ug-integration", "version": "1.0.0"}}, + } + ) + + "\n" + ) + proc.stdin.flush() + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + try: + message = messages.get(timeout=max(0.01, deadline - time.monotonic())) + except queue.Empty: + break + if message is None: + break + assert isinstance(message, dict), message + assert "protocol_error" not in message, ( + "Non-JSON output on the app-server protocol stream: " + str(message) + ) + if isinstance(message, dict) and message.get("id") == 1: + assert "error" not in message, message + assert isinstance(message.get("result"), dict), message + assert message["result"].get("userAgent"), message + proc.stdin.write('{"method":"initialized","params":{}}\n') + proc.stdin.flush() + return message + raise AssertionError("No app-server initialize response:\n" + "".join(transcript)) + finally: + stop_process(proc) + reader.join(timeout=5) + stderr_reader.join(timeout=5) + proc.stdin.close() + proc.stdout.close() + proc.stderr.close() + self.record( + "app-server.json", {"argv": command, "stdout": transcript, "stderr": diagnostics} + ) diff --git a/tests/integration/utils/terminal.py b/tests/integration/utils/terminal.py new file mode 100644 index 000000000..ea6b05208 --- /dev/null +++ b/tests/integration/utils/terminal.py @@ -0,0 +1,294 @@ +"""A real PTY and terminal screen for driving installed interactive agents. + +The screen implements terminal device replies; it never substitutes an agent, +gateway, application function, configuration file, or model response. +""" + +from __future__ import annotations + +import contextlib +import os +import re +import signal +import time +import uuid + +import pexpect +import pyte + +from .evidence import agent_sessions + + +class TerminalScreen(pyte.Screen): + def __init__(self, columns, lines, send): + super().__init__(columns, lines) + self.send = send + + def write_process_input(self, data): + # Real TUIs query cursor position/device attributes during startup. + # pyte replies according to the terminal state it has actually rendered. + self.send(data) + + +class TerminalProcess: + def __init__(self, session, agent, command, name): + self.session = session + self.agent = agent + self.name = name + self.command = command + env = {**session.env, "TERM": "xterm-256color"} + self.child = pexpect.spawn( + self.command[0], + self.command[1:], + cwd=str(session.cwd), + env=env, + encoding="utf-8", + codec_errors="replace", + dimensions=(60, 140), + timeout=120, + ) + self.screen = TerminalScreen(140, 60, self.child.send) + self.stream = pyte.Stream(self.screen) + self.output = [] + self.actions = [] + self.ended = False + + @property + def visible(self): + return "\n".join(line.rstrip() for line in self.screen.display) + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc, traceback): + # pexpect creates a new session with a controlling terminal. Clean up + # its process group even if the ug leader exited before its children. + with contextlib.suppress(ProcessLookupError): + os.killpg(self.child.pid, signal.SIGTERM) + self.child.close(force=True) + with contextlib.suppress(ProcessLookupError): + os.killpg(self.child.pid, signal.SIGKILL) + routing_log = ( + self.session.home + / ".ucode" + / ("claude-v2-pty.log" if self.agent == "claude" else "codex-v2-interposer.log") + ) + self.session.record( + f"{self.name}.json", + { + "argv": self.command, + "terminal": {"rows": 60, "columns": 140, "term": "xterm-256color"}, + "actions": self.actions, + "transcript": "".join(self.output), + "screen": self.visible, + "exitstatus": self.child.exitstatus, + "signalstatus": self.child.signalstatus, + "normal_exit_observed": self.ended and self.child.exitstatus == 0, + "routing_log": routing_log.read_text() if routing_log.is_file() else None, + "agent_sessions": agent_sessions(self.session, self.agent) + if self.agent in ("claude", "codex") + else {}, + }, + ) + + def read(self): + try: + chunk = self.child.read_nonblocking(size=65536, timeout=0.2) + except pexpect.TIMEOUT: + return + except pexpect.EOF: + self.ended = True + return + self.output.append(chunk) + self.stream.feed(chunk) + + def send(self, keys, reason): + self.actions.append({"reason": reason, "keys": keys, "screen_before": self.visible}) + self.child.send(keys) + + def wait_for(self, predicate, description, timeout=30, stable_for=0.3): + deadline = time.monotonic() + timeout + since = None + while time.monotonic() < deadline: + self.read() + assert not self.ended, f"TUI exited while waiting for {description}:\n{self.visible}" + if predicate(self.visible): + since = since or time.monotonic() + if time.monotonic() - since >= stable_for: + return + else: + since = None + raise AssertionError(f"TUI did not show {description} within {timeout}s:\n{self.visible}") + + def selected_line(self): + return next( + (line.strip() for line in self.visible.splitlines() if re.match(r"^\s*[›❯>]", line)), + "", + ) + + def choose(self, prompt, label): + """Navigate the visible menu with arrow keys; never write its saved state.""" + self.wait_for(lambda text: prompt in text and self.selected_line(), prompt, timeout=120) + visited = set() + for _ in range(100): + current = self.selected_line() + if label in current: + self.send("\r", f"choose {label}") + self.wait_for( + lambda text, before=current: ( + prompt not in text or self.selected_line() != before + ), + f"confirmation of {label}", + ) + return + assert current not in visited, f"Menu does not offer {label}:\n{self.visible}" + visited.add(current) + self.send("\x1b[B", f"move towards {label}") + self.wait_for( + lambda text, before=current: self.selected_line() != before, "next menu option" + ) + raise AssertionError(f"Could not select {label}") + + def submit(self, text): + self.send(text, "type text before pressing Enter") + compact = "".join(text.split()) + self.wait_for( + lambda screen: compact in "".join(screen.split()), "typed input", stable_for=0.5 + ) + self.send("\r", "press Enter after the input has rendered") + + def finish(self, timeout=120): + deadline = time.monotonic() + timeout + while not self.ended and time.monotonic() < deadline: + self.read() + assert self.ended, f"Process did not exit within {timeout}s:\n{self.visible}" + self.child.close(force=False) + assert self.child.exitstatus == 0, ( + f"exit={self.child.exitstatus}, signal={self.child.signalstatus}:\n{self.visible}" + ) + + +class ConfigureTerminal(TerminalProcess): + def select_agent(self, display): + prompt = "Select coding agents to configure:" + self.wait_for(lambda text: prompt in text and self.selected_line(), prompt, timeout=120) + visited = set() + selected = False + while True: + current = self.selected_line() + match = re.search(r"[›❯>]\s*([●○])\s*(.+)", current) + assert match, f"Unrecognized agent checkbox: {current}" + checked, name = match.groups() + if name in visited: + break + visited.add(name) + wanted = name.strip() == display + selected = selected or wanted + if (checked == "●") != wanted: + self.send(" ", f"{'select' if wanted else 'deselect'} {name}") + self.wait_for( + lambda text, before=current: self.selected_line() != before, "checkbox change" + ) + current = self.selected_line() + self.send("\x1b[B", "next agent checkbox") + self.wait_for(lambda text, before=current: self.selected_line() != before, "next agent") + assert selected, f"{display} was not available in the real agent picker" + self.send("\r", f"configure only {display}") + + +class AgentTerminal(TerminalProcess): + def boot(self, timeout=120): + """Handle only recognized visible onboarding; unknown screens fail.""" + handled = set() + ready_since = None + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + self.read() + assert not self.ended, f"TUI exited before its prompt:\n{self.visible}" + text = self.visible + if "Accessing workspace:" in text and self.session.cwd.name in text: + self.choose("Accessing workspace:", "Yes, I trust this folder") + continue + if "Hooks need review" in text: + self.choose("Hooks need review", "Trust all and continue") + continue + # These are interactions with ordinary UI choices, not pre-written + # onboarding state. Only the test's disposable project is trusted. + dialogs = [ + ( + "theme", + "Choose the text style" in text and "Dark mode" in text, + "\r", + ), + ( + "security-notes", + "Security notes" in text and "Enter to continue" in text, + "\r", + ), + ( + "trust-folder", + self.session.cwd.name in text + and bool(re.search(r"1[.)]\s+Yes, I trust (?:this|the) folder", text)), + "\r", + ), + ( + "trust-directory", + self.session.cwd.name in text + and "trust" in text.lower() + and bool(re.search(r"1[.)]\s+Yes, (?:continue|proceed)", text)), + "\r", + ), + ] + matched = False + for label, shown, keys in dialogs: + if shown: + matched = True + if label not in handled: + self.send(keys, label) + handled.add(label) + break + if matched: + continue + assert "Select login method:" not in text, ( + "Configured ug launched Claude's account-login flow instead of its gateway session:\n" + + text + ) + assert not ("Sign in with ChatGPT" in text and "Provide your own API key" in text), ( + "Configured ug launched Codex's account-login flow instead of its gateway session:\n" + + text + ) + title = "Claude Code" if self.agent == "claude" else "Codex" + if ( + title in text + and "loading" not in text.lower() + and re.search(r"(?m)^\s*[❯›>]\s*(?!\d+[.)])", text) + ): + ready_since = ready_since or time.monotonic() + if time.monotonic() - ready_since >= 1: + self.actions.append({"reason": "prompt-ready", "screen": text}) + return + else: + ready_since = None + raise AssertionError( + f"TUI did not reach a usable prompt within {timeout}s:\n{self.visible}" + ) + + def check_input_and_exit(self): + marker = "ug-boot-" + uuid.uuid4().hex[:12] + self.send(marker, "type into the prompt without submitting a model request") + self.wait_for(lambda text: marker in text, "typed text in the prompt") + self.send("\x15", "Ctrl-U clears the prompt") + self.wait_for(lambda text: marker not in text, "cleared prompt") + self.exit_normally() + + def wait_for_task(self, task, timeout=180): + self.wait_for( + lambda text: task.completed(self.session, self.agent), + "a completed assistant answer with the file's value", + timeout=timeout, + ) + task.assert_completed(self.session, self.agent) + + def exit_normally(self): + self.submit("/exit") + self.finish(timeout=30) diff --git a/tests/test_agent_claude.py b/tests/test_agent_claude.py index 23826da34..4a8a0ac14 100644 --- a/tests/test_agent_claude.py +++ b/tests/test_agent_claude.py @@ -1727,3 +1727,40 @@ def test_missing_login_runs_browser_flow(self, monkeypatch): monkeypatch.setattr(claude, "print_success", lambda *a, **kw: None) claude._ensure_subscription_login() assert calls == [[claude.SPEC["binary"], "auth", "login"]] + + +class TestWriteToolConfigBackup: + """A re-configure must not snapshot the file ucode itself generated.""" + + def _patch(self, monkeypatch, tmp_path): + monkeypatch.setattr(claude, "CLAUDE_SETTINGS_PATH", tmp_path / "ucode-settings.json") + monkeypatch.setattr(claude, "CLAUDE_BACKUP_PATH", tmp_path / "backup.json") + monkeypatch.setattr("ucode.config_io.APP_DIR", tmp_path) + monkeypatch.setattr(claude, "save_state", lambda state: None) + monkeypatch.setattr(claude, "_register_web_search_mcp", lambda *a, **kw: None) + + def test_first_configure_backs_up_user_owned_file(self, tmp_path, monkeypatch): + self._patch(monkeypatch, tmp_path) + (tmp_path / "ucode-settings.json").write_text( + '{"permissions": {"allow": ["Read"]}}', encoding="utf-8" + ) + + claude.write_tool_config( + {"workspace": WS, "claude_models": {}}, "databricks-claude-sonnet-4" + ) + + backup = (tmp_path / "backup.json").read_text(encoding="utf-8") + assert backup == '{"permissions": {"allow": ["Read"]}}' + + def test_reconfigure_does_not_back_up_generated_file(self, tmp_path, monkeypatch): + self._patch(monkeypatch, tmp_path) + state = { + "workspace": WS, + "claude_models": {}, + # load_state after a first configure: ucode already manages this file. + "managed_configs": {"claude": {"keys": [["env", "ANTHROPIC_BASE_URL"]]}}, + } + + claude.write_tool_config(state, "databricks-claude-sonnet-4") + + assert not (tmp_path / "backup.json").exists() diff --git a/tests/test_agent_codex.py b/tests/test_agent_codex.py index 07162a5fa..7185cfd20 100644 --- a/tests/test_agent_codex.py +++ b/tests/test_agent_codex.py @@ -1065,3 +1065,49 @@ def deny_managed_write(*args, **kwargs): with pytest.raises(managed_files.ManagedFileWriteUnavailable, match="sudo denied"): codex.write_tool_config({"workspace": WS, "codex_models": ["gpt-5"]}) + + +class TestWriteConfigBackup: + """A re-configure must not snapshot the file ucode itself generated.""" + + def _patch(self, monkeypatch, tmp_path): + monkeypatch.setattr(codex, "CODEX_CONFIG_PATH", tmp_path / "ucode.config.toml") + monkeypatch.setattr(codex, "CODEX_BACKUP_PATH", tmp_path / "backup.toml") + monkeypatch.setattr("ucode.config_io.APP_DIR", tmp_path) + monkeypatch.setattr(codex, "agent_version", lambda binary: "0.134.0") + monkeypatch.setattr(codex, "save_state", lambda state: None) + + def test_first_configure_backs_up_user_owned_profile(self, tmp_path, monkeypatch): + self._patch(monkeypatch, tmp_path) + (tmp_path / "ucode.config.toml").write_text("# user comment\n", encoding="utf-8") + + codex.write_tool_config({"workspace": WS, "codex_models": ["gpt-5"]}) + + assert (tmp_path / "backup.toml").read_text(encoding="utf-8") == "# user comment\n" + + def test_reconfigure_does_not_back_up_generated_profile(self, tmp_path, monkeypatch): + self._patch(monkeypatch, tmp_path) + state = { + "workspace": WS, + "codex_models": ["gpt-5"], + # load_state after a first configure: ucode already manages this file. + "managed_configs": {"codex": {"keys": [["model_provider"]]}}, + } + + codex.write_tool_config(state) + + assert not (tmp_path / "backup.toml").exists() + + def test_clear_model_preferences_does_not_back_up_generated_profile( + self, tmp_path, monkeypatch + ): + self._patch(monkeypatch, tmp_path) + (tmp_path / "ucode.config.toml").write_text('model = "system.ai.gpt-5"\n', encoding="utf-8") + + changed = codex.clear_model_preferences( + {"workspace": WS, "managed_configs": {"codex": {"keys": []}}} + ) + + assert changed is True + assert "model" not in read_toml_safe(tmp_path / "ucode.config.toml") + assert not (tmp_path / "backup.toml").exists() diff --git a/tests/test_cli.py b/tests/test_cli.py index 65638e89f..dddd90308 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -4024,3 +4024,28 @@ def test_still_logs_in_when_nothing_external_is_set(self, monkeypatch): monkeypatch.delenv("DATABRICKS_BEARER_COMMAND", raising=False) assert self._run(monkeypatch) == ["https://ws.cloud.databricks.com"] + + +class TestStdioProtocolLaunch: + """`codex app-server` owns stdout, so ug's status output moves to stderr.""" + + def test_app_server_subcommand_owns_stdout(self): + assert cli_mod._child_owns_stdout("codex", ["app-server", "--listen", "stdio://"]) is True + + def test_other_codex_launches_keep_stdout(self): + assert cli_mod._child_owns_stdout("codex", []) is False + assert cli_mod._child_owns_stdout("codex", ["exec", "--json", "hi"]) is False + + def test_other_agents_never_own_stdout(self): + assert cli_mod._child_owns_stdout("claude", ["app-server"]) is False + assert cli_mod._child_owns_stdout("gemini", []) is False + + def test_redirect_rebinds_stdout_without_touching_the_descriptor(self): + import sys + + real_stdout = sys.stdout + try: + cli_mod.redirect_output_to_stderr() + assert sys.stdout is sys.stderr + finally: + sys.stdout = real_stdout diff --git a/tests/test_integration_contract.py b/tests/test_integration_contract.py new file mode 100644 index 000000000..a637b9923 --- /dev/null +++ b/tests/test_integration_contract.py @@ -0,0 +1,54 @@ +"""Keep the black-box suite independent of application internals and test doubles.""" + +import ast +from pathlib import Path + + +def test_integration_suite_uses_only_public_process_boundaries(): + violations = [] + for path in (Path(__file__).parent / "integration").rglob("*.py"): + for node in ast.walk(ast.parse(path.read_text())): + modules = [] + if isinstance(node, ast.Import): + modules = [alias.name for alias in node.names] + elif isinstance(node, ast.ImportFrom): + modules = [node.module or ""] + if any(module.split(".")[0] in {"ucode", "mock", "unittest"} for module in modules): + violations.append(f"{path.name}:{node.lineno}: imports application or test doubles") + if isinstance(node, ast.Name) and node.id in { + "monkeypatch", + "MonkeyPatch", + "Mock", + "MagicMock", + "patch", + "setattr", + "delattr", + }: + violations.append(f"{path.name}:{node.lineno}: uses {node.id}") + if isinstance(node, ast.Attribute) and node.attr in { + "MonkeyPatch", + "Mock", + "MagicMock", + "mock", + "patch", + "skip", + "skipif", + "xfail", + }: + violations.append(f"{path.name}:{node.lineno}: uses {node.attr}") + assert not violations, "\n".join(violations) + + +def test_integration_tests_describe_the_scenario_and_expected_result(): + root = Path(__file__).parent / "integration" + violations = [] + for path in root.rglob("test_*.py"): + for node in ast.walk(ast.parse(path.read_text())): + if not isinstance(node, ast.FunctionDef) or not node.name.startswith("test_"): + continue + description = ast.get_docstring(node) or "" + if "Scenario:" not in description or "Expected:" not in description: + violations.append(f"{path.name}:{node.lineno}: describe Scenario and Expected") + if any(arg.arg == "configured" for arg in node.args.args): + violations.append(f"{path.name}:{node.lineno}: setup must be visible in the test") + assert not violations, "\n".join(violations)