From 93db13590e673e08f2d4c71720bcda20095d2d0e Mon Sep 17 00:00:00 2001 From: Erdenebileg Bat-Amgalan Date: Sun, 6 Sep 2026 21:09:30 +0900 Subject: [PATCH] feat(qa): add declarative verification harness --- .agents/DECISIONS.md | 1 + .agents/PROGRESS.md | 4 + .agents/agent-harness/README.md | 14 ++ .../decisions/0017-declarative-qa-profiles.md | 22 +++ .agents/decisions/README.md | 1 + .agents/tasks/qa-harness-integration.md | 56 ++++++ .gitignore | 2 + Makefile | 7 +- qa/README.md | 55 ++++++ qa/qa.config.json | 52 ++++++ qa/runner/harness.py | 164 ++++++++++++++++++ qa/runner/harness_lib.py | 121 +++++++++++++ qa/runner/test_harness.py | 54 ++++++ qa/runner/test_harness_lib.py | 34 ++++ 14 files changed, 585 insertions(+), 2 deletions(-) create mode 100644 .agents/decisions/0017-declarative-qa-profiles.md create mode 100644 .agents/tasks/qa-harness-integration.md create mode 100644 qa/README.md create mode 100644 qa/qa.config.json create mode 100644 qa/runner/harness.py create mode 100644 qa/runner/harness_lib.py create mode 100644 qa/runner/test_harness.py create mode 100644 qa/runner/test_harness_lib.py diff --git a/.agents/DECISIONS.md b/.agents/DECISIONS.md index b0b9cc5..ec86cae 100644 --- a/.agents/DECISIONS.md +++ b/.agents/DECISIONS.md @@ -17,3 +17,4 @@ | Provider-neutral agent protocol, Paseo boundary, and local runtime release | Accepted | `.agents/decisions/0013-agent-protocol-and-local-runtime-release.md` | | Stable latest-release Unix installer asset | Accepted | `.agents/decisions/0015-stable-latest-release-installer.md` | | Local runtime wiring without embedded scheduling | Accepted | `.agents/decisions/0016-runtime-wiring-boundary.md` | +| Declarative QA profiles around the canonical gate | Accepted | `.agents/decisions/0017-declarative-qa-profiles.md` | diff --git a/.agents/PROGRESS.md b/.agents/PROGRESS.md index 9f299cb..67354a1 100644 --- a/.agents/PROGRESS.md +++ b/.agents/PROGRESS.md @@ -9,6 +9,10 @@ Phase 9 — runtime wiring in progress. - Defined product goal, normative domain model, and implementation roadmap. - Resolved revision identity, canonical content, exact composition, review targeting, and integration semantics. - Established the evidence-driven development, review, CI, and specification-release harness. +- Added a Weft-specific declarative QA harness with dependency-aware profiles, + bounded reports, explicit PASS/FAIL/BLOCKED/SKIPPED evidence, process-group + timeout cleanup, release archive smoke composition, and optional live-provider + observations; it preserves `make check` as the canonical gate. - Added a passing Native Git Phase 0 spike for canonical reconstruction, provider rewrite survival, candidate composition, target guarding, conflict capture, and external-ref reconciliation. diff --git a/.agents/agent-harness/README.md b/.agents/agent-harness/README.md index bc9d94e..47b7ba2 100644 --- a/.agents/agent-harness/README.md +++ b/.agents/agent-harness/README.md @@ -29,6 +29,20 @@ Provider names, branches, commands, and successful happy paths do not prove iden - Add a project skill only for a repeated workflow with deterministic inputs, outputs, and validation. - Keep historical task reports out of normative specifications. +## Declarative QA profiles + +`qa/qa.config.json` declares additive QA profiles around existing Weft commands; +it does not replace `make check`. `python3 qa/runner/harness.py` runs the default +smoke profile, resolves each suite's dependencies once, runs commands without a +shell, terminates timed-out process groups, and writes ignored bounded reports +under `qa/reports/`. Results are explicit: `PASS` ran successfully, `FAIL` ran +and failed, `BLOCKED` lacks a mandatory local capability, and `SKIPPED` is +inapplicable or depends on unsatisfied proof. `full` adds Native Git feasibility, +`release --version vMAJOR.MINOR.PATCH` builds and smoke-tests the matching archive, +and `provider-observation` exposes version-gated GitButler evidence without +claiming it is deterministic. The runner's unit/lifecycle self-test is part of +`make check`. + ## Orchestrator boundary Agent runtimes use the provider-neutral process contract in diff --git a/.agents/decisions/0017-declarative-qa-profiles.md b/.agents/decisions/0017-declarative-qa-profiles.md new file mode 100644 index 0000000..647d557 --- /dev/null +++ b/.agents/decisions/0017-declarative-qa-profiles.md @@ -0,0 +1,22 @@ +# ADR-0017: Declarative QA profiles around the canonical gate + +- **Status:** Accepted +- **Date:** 2026-09-06 + +## Decision + +Weft adopts a small standard-library QA runner with declarative smoke, full, +release, and opt-in provider-observation profiles. It resolves suite dependencies, +preflights local capabilities, bounds output, cleans up timed-out process groups, +emits ignored reports, and distinguishes pass, failure, blocking, and skipped +coverage. `make check` remains the canonical local and CI gate; the QA runner +does not replace or recursively redefine it. + +## Consequences + +- QA configuration contains only Rust/CLI/release-relevant checks. +- Generated reports remain local under `qa/reports/`. +- `SKIPPED` is disclosed evidence, never a passing provider/platform claim. +- DBMS, CMake, compatibility-container, and branch-name rules from the source + project are deliberately excluded. +- The runner self-test is included in `make check`. diff --git a/.agents/decisions/README.md b/.agents/decisions/README.md index f0b9a96..038518c 100644 --- a/.agents/decisions/README.md +++ b/.agents/decisions/README.md @@ -22,3 +22,4 @@ Each ADR records status, context, decision, alternatives, consequences, migratio - [ADR-0014: Minimal runtime archive and release metadata](0014-minimal-runtime-archive-and-release-metadata.md) - [ADR-0015: Stable latest-release Unix installer asset](0015-stable-latest-release-installer.md) - [ADR-0016: Local runtime wiring without embedded scheduling](0016-runtime-wiring-boundary.md) +- [ADR-0017: Declarative QA profiles around the canonical gate](0017-declarative-qa-profiles.md) diff --git a/.agents/tasks/qa-harness-integration.md b/.agents/tasks/qa-harness-integration.md new file mode 100644 index 0000000..7adcb6a --- /dev/null +++ b/.agents/tasks/qa-harness-integration.md @@ -0,0 +1,56 @@ +# Task Record: Weft declarative QA harness integration + +## Outcome and scope + +- **User/operator result:** Weft has a local, declarative QA entrypoint that + composes existing proof commands and reports their exact outcome without + replacing the repository gate. +- **In scope:** Dependency-aware suite planning, preflight capability checks, + bounded reports, timeout cleanup, stable result/exit semantics, runner tests, + and Weft-specific smoke/full/release profiles. +- **Out of scope:** Agent scheduling, provider mutation, CMake/DBMS/container + policies from the source project, and claiming unavailable platform coverage. +- **Affected domain invariants:** None directly; harness evidence must never + overstate provider, release, or recovery proof. +- **Provider/runtime scope:** Local Rust CLI, Native Git feasibility evidence, + optional GitButler live evidence, and release archives. +- **Compatibility surface:** CLI, harness, release. + +## Acceptance criteria + +1. Profiles resolve dependency order once and fail safely on invalid config or cycles. +2. Every suite reports PASS, FAIL, BLOCKED, or SKIPPED with stable exit behavior. +3. Commands run directly (without a shell), time out as a process group, and retain bounded output. +4. `make check` validates the runner contract; smoke and release profiles are exercised end to end. + +## Risks + +- **Data/security:** Commands must not interpolate through a shell or capture unbounded output. +- **Concurrency/crash recovery:** Timeout must terminate suite descendants; reports must disclose interruption/blocking. +- **Provider divergence/compatibility:** Optional provider/platform evidence remains skipped, never passed. +- **Upgrade/rollback:** Release profile uses the declared version and existing archive proof. + +## Evidence and plan + +- Relevant paths: `qa/`, `Makefile`, `.agents/agent-harness/`, `docs/DEPLOYMENT.md`. +- Source comparison: retain declarative planning/status/reporting from ezis-nexus; + exclude CMake, DBMS, compatibility-container, and branch-target rules. + +1. Build pure config/planning/report helpers — unit tests for invalid configs, ordering, and exit codes. +2. Build Weft runner lifecycle — tests for preflight, dependency skip, and timeout cleanup. +3. Define Weft profiles — end-to-end smoke/release evidence and full local gate. + +## Validation record + +| Check | Command/test | Result | Evidence | +| --- | --- | --- | --- | +| Focused | `python3 qa/runner/harness.py self-test` | Passed | Config, dependency, status, preflight, and descendant-timeout tests | +| Harness/docs | `python3 qa/runner/harness.py --profile smoke` | Passed | Canonical gate executed through declarative profile | +| Release | `python3 qa/runner/harness.py --profile release --version v0.2.1` | Passed | Gate, package, clean archive install/restart/uninstall | +| Provider observation | `python3 qa/runner/harness.py --profile provider-observation` | Passed | Explicit live GitButler adapter proof in this environment | + +## Decision and follow-up + +- **Decision and alternatives rejected:** Retain additive orchestration, not a replacement build system or agent scheduler. +- **Residual risks:** Optional live-provider evidence remains environment-dependent and explicitly reported; it is not part of `make check`. +- **Follow-up:** Add a suite only when its proof contract is stable and directly relevant to Weft. diff --git a/.gitignore b/.gitignore index 6ef8608..b54fc8d 100644 --- a/.gitignore +++ b/.gitignore @@ -17,6 +17,8 @@ tmp/ build/ dist/ coverage/ +qa/reports/ +__pycache__/ target/ .env .env.* diff --git a/Makefile b/Makefile index 8f58b99..6f2c991 100644 --- a/Makefile +++ b/Makefile @@ -1,8 +1,8 @@ -.PHONY: check docs-check harness-check rust-check phase0-spike phase0-native-git-spike phase0-gitbutler-spike phase3-gitbutler-live package-release test-release test-upgrade-rollback test-release-reproducibility +.PHONY: check docs-check harness-check qa-self-test rust-check phase0-spike phase0-native-git-spike phase0-gitbutler-spike phase3-gitbutler-live package-release test-release test-upgrade-rollback test-release-reproducibility RUST_HOST := $(shell rustc -vV | sed -n 's/^host: //p') -check: harness-check docs-check rust-check +check: harness-check docs-check qa-self-test rust-check harness-check: ./scripts/verify-harness.sh @@ -10,6 +10,9 @@ harness-check: docs-check: python3 scripts/check_docs.py +qa-self-test: + python3 qa/runner/harness.py self-test + rust-check: cargo fmt --all --check cargo test --workspace --all-targets --target $(RUST_HOST) diff --git a/qa/README.md b/qa/README.md new file mode 100644 index 0000000..109497e --- /dev/null +++ b/qa/README.md @@ -0,0 +1,55 @@ +# Weft QA Harness + +The QA harness is a small standard-library orchestration layer around Weft's +existing Make and script entrypoints. It does not replace `make check`, provider +tests, or release scripts; it declares when and how those existing proofs are +combined, then produces bounded local evidence. + +## Commands + +```bash +# Canonical deterministic repository gate +python3 qa/runner/harness.py --profile smoke + +# Add Native Git provider feasibility evidence +python3 qa/runner/harness.py --profile full + +# Gate, package, and clean archive install/restart/uninstall proof +python3 qa/runner/harness.py --profile release --version v0.2.1 + +# Explicit, environment-dependent GitButler adapter observation +python3 qa/runner/harness.py --profile provider-observation + +# Runner configuration, planning, lifecycle, timeout, and report tests +python3 qa/runner/harness.py self-test +``` + +Profiles and suites live in `qa.config.json`. Commands execute as argument arrays, +not shell strings. Dependencies resolve once in order; a dependent suite is +`SKIPPED` if its prerequisite did not establish the required evidence. + +## Result contract + +| Status | Meaning | +| --- | --- | +| `PASS` | The declared command ran and completed successfully. | +| `FAIL` | The command ran but failed or exceeded its deadline. | +| `BLOCKED` | A required option or mandatory local capability is unavailable. | +| `SKIPPED` | The suite is inapplicable, optional capability is unavailable, or a dependency was not satisfied. | + +Exit code `0` means no executed suite failed and at least one proof passed; `1` +means an executed suite failed; `2` means blocked, all-skipped, or invalid +configuration. A skipped provider/platform suite is disclosure, not coverage. + +Reports are written to the ignored `qa/reports//` directory. `latest.json` +is machine-readable and the timestamped Markdown companion is human-readable. +Output tails are capped at 24,000 characters. On POSIX, a timeout terminates the +suite process group so descendants cannot outlive the result. + +## Weft-specific boundaries + +The harness adopts declarative planning, lifecycle status, bounded reporting, and +standard-library portability from ezis-nexus. It deliberately excludes that +project's CMake build assumptions, DBMS branch rules, compatibility containers, +and self-hosted runner checks. It never launches or schedules agents; Weft's +durable agent/process boundary remains defined in `.agents/AGENT_PROTOCOL.md`. diff --git a/qa/qa.config.json b/qa/qa.config.json new file mode 100644 index 0000000..0bd622f --- /dev/null +++ b/qa/qa.config.json @@ -0,0 +1,52 @@ +{ + "schemaVersion": 1, + "profiles": { + "smoke": ["repository-gate"], + "full": ["repository-gate", "native-git-spike"], + "release": ["repository-gate", "release-archive-smoke"], + "provider-observation": ["gitbutler-live"] + }, + "suites": { + "repository-gate": { + "description": "Run the canonical local repository gate", + "kind": "deterministic", + "command": ["make", "check"], + "requiredCommands": ["make"], + "timeoutSeconds": 900 + }, + "native-git-spike": { + "description": "Reproduce the native Git feasibility evidence", + "kind": "deterministic", + "command": ["make", "phase0-native-git-spike"], + "requiredCommands": ["make", "git"], + "timeoutSeconds": 300 + }, + "package-release": { + "description": "Build the release archive for the declared version", + "kind": "deterministic", + "command": ["make", "package-release", "VERSION={version}"], + "requiresVariables": ["version"], + "requiredCommands": ["make"], + "timeoutSeconds": 900 + }, + "release-archive-smoke": { + "description": "Install, restart, and uninstall the declared release archive", + "kind": "deterministic", + "requires": ["package-release"], + "requiresVariables": ["version"], + "command": ["make", "test-release", "ARCHIVE=dist/weft-{release_version}-x86_64-unknown-linux-musl.tar.gz"], + "requiredCommands": ["make"], + "platforms": ["linux"], + "timeoutSeconds": 300 + }, + "gitbutler-live": { + "description": "Observe the explicitly version-gated GitButler adapter proof", + "kind": "observation", + "optional": true, + "command": ["make", "phase3-gitbutler-live"], + "requiredCommands": ["make", "but"], + "platforms": ["linux"], + "timeoutSeconds": 300 + } + } +} diff --git a/qa/runner/harness.py b/qa/runner/harness.py new file mode 100644 index 0000000..443fc04 --- /dev/null +++ b/qa/runner/harness.py @@ -0,0 +1,164 @@ +#!/usr/bin/env python3 +"""Weft's additive, declarative local QA runner.""" + +from __future__ import annotations + +import argparse +import datetime as dt +import os +import platform +import shutil +import signal +import subprocess +import sys +import time +from pathlib import Path +from typing import Any, Mapping + +from harness_lib import HarnessConfigError, exit_code_for, load_config as load_validated_config, render_command, resolve_plan, summarize_results, write_report as write_structured_report + +ROOT = Path(__file__).resolve().parents[2] +CONFIG = ROOT / "qa" / "qa.config.json" +OUTPUT_LIMIT = 24_000 + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description="Run declarative Weft QA profiles.") + parser.add_argument("action", choices=("run", "list", "self-test"), nargs="?", default="run") + parser.add_argument("--profile", default="smoke") + parser.add_argument("--suite", action="append", default=[]) + parser.add_argument("--version", default="") + parser.add_argument("--keep-going", action="store_true") + parser.add_argument("--report-dir", default="qa/reports") + return parser.parse_args() + + +def load_config() -> dict: + return load_validated_config(CONFIG) + + +def command_for(suite: Mapping[str, Any], variables: Mapping[str, str]) -> list[str]: + missing = [name for name in suite.get("requiresVariables", []) if not variables.get(name)] + if missing: + raise HarnessConfigError(f"missing required options: {', '.join('--' + name for name in missing)}") + return render_command(suite["command"], variables) + + +def _terminate(process: subprocess.Popen[str]) -> None: + if process.poll() is not None: + return + try: + if os.name == "posix": + os.killpg(process.pid, signal.SIGTERM) + else: + process.terminate() + process.wait(timeout=5) + except (ProcessLookupError, subprocess.TimeoutExpired): + if process.poll() is None: + if os.name == "posix": + os.killpg(process.pid, signal.SIGKILL) + else: + process.kill() + + +def _preflight(suite: Mapping[str, Any], variables: Mapping[str, str]) -> tuple[str, str] | None: + if suite.get("platforms") and platform.system().lower() not in suite["platforms"]: + return "SKIPPED", f"platform {platform.system().lower()} is not applicable" + required = [render_command([item], variables)[0] for item in suite.get("requiredCommands", [])] + missing = [item for item in required if shutil.which(item) is None] + if missing: + return ("SKIPPED" if suite.get("optional") else "BLOCKED", f"missing required commands: {', '.join(missing)}") + alternatives = [render_command([item], variables)[0] for item in suite.get("requiredAnyCommands", [])] + if alternatives and not any(shutil.which(item) for item in alternatives): + return ("SKIPPED" if suite.get("optional") else "BLOCKED", f"no supported command available: {', '.join(alternatives)}") + return None + + +def run_suite(name: str, suite: Mapping[str, Any], variables: Mapping[str, str]) -> dict: + started = time.monotonic() + try: + command = command_for(suite, variables) + except HarnessConfigError as error: + return {"name": name, "kind": suite.get("kind", "deterministic"), "status": "BLOCKED", "satisfiesDependencies": False, "note": str(error), "durationMs": 0} + preflight = _preflight(suite, variables) + if preflight: + return {"name": name, "kind": suite.get("kind", "deterministic"), "status": preflight[0], "satisfiesDependencies": False, "note": preflight[1], "durationMs": 0} + try: + process = subprocess.Popen(command, cwd=ROOT, text=True, errors="replace", stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, start_new_session=(os.name == "posix")) + output, _ = process.communicate(timeout=suite.get("timeoutSeconds", 300)) + if process.returncode == 0: + status, note = "PASS", "" + elif process.returncode in suite.get("skipExitCodes", []): + status, note = "SKIPPED", f"unavailable coverage (exit {process.returncode})" + else: + status, note = "FAIL", f"command exited with {process.returncode}" + exit_code = process.returncode + except OSError as error: + return {"name": name, "kind": suite.get("kind", "deterministic"), "status": "BLOCKED", "satisfiesDependencies": False, "note": f"cannot start command: {error}", "durationMs": round((time.monotonic() - started) * 1000)} + except subprocess.TimeoutExpired: + _terminate(process) + output, _ = process.communicate() + status, note, exit_code = "FAIL", f"timed out after {suite.get('timeoutSeconds', 300)} seconds", 124 + except KeyboardInterrupt: + _terminate(process) + output, _ = process.communicate() + status, note, exit_code = "BLOCKED", "interrupted by user", 143 + return {"name": name, "kind": suite.get("kind", "deterministic"), "status": status, + "satisfiesDependencies": status == "PASS", "note": note, "command": command, + "exitCode": exit_code, "durationMs": round((time.monotonic() - started) * 1000), "outputTail": output[-OUTPUT_LIMIT:]} + + +def self_test() -> int: + return subprocess.run( + [sys.executable, "-m", "unittest", "discover", "-s", "qa/runner", "-p", "test_*.py", "-v"], + cwd=ROOT, + check=False, + ).returncode + + +def main() -> int: + args = parse_args() + try: + config = load_config() + if args.action == "self-test": + return self_test() + if args.action == "list": + for name, suites in config["profiles"].items(): + print(f"{name}: {', '.join(suites)}") + return 0 + selected = args.suite or config["profiles"].get(args.profile, []) + if not selected: + raise HarnessConfigError(f"unknown or empty QA profile: {args.profile}") + plan = resolve_plan(config, selected) + variables = {"version": args.version, "release_version": args.version.removeprefix("v")} + results = [] + satisfied: dict[str, bool] = {} + stop_reason = "" + for item in plan: + missing = [name for name in item.config.get("requires", []) if not satisfied.get(name, False)] + if missing: + result = {"name": item.name, "kind": item.config.get("kind", "deterministic"), "status": "SKIPPED", "satisfiesDependencies": False, "note": f"dependency did not satisfy prerequisites: {', '.join(missing)}", "durationMs": 0} + elif stop_reason: + result = {"name": item.name, "kind": item.config.get("kind", "deterministic"), "status": "SKIPPED", "satisfiesDependencies": False, "note": stop_reason, "durationMs": 0} + else: + result = run_suite(item.name, item.config, variables) + results.append(result) + satisfied[item.name] = bool(result["satisfiesDependencies"]) + print(f"{item.name}: {result['status']} ({result['note']})") + if result["status"] in ("FAIL", "BLOCKED") and not args.keep_going: + stop_reason = f"stopped after {item.name} {result['status'].lower()}" + summary = summarize_results(results) + report = {"schemaVersion": 1, "runLabel": "custom" if args.suite else args.profile, + "startedAt": dt.datetime.now(dt.UTC).isoformat(timespec="milliseconds").replace("+00:00", "Z"), + "selectedSuites": selected, "executionPlan": [item.name for item in plan], + "summary": summary, "results": results} + write_structured_report(report, ROOT / args.report_dir) + return exit_code_for(summary) + except HarnessConfigError as error: + print(f"HARNESS ERROR: {error}", file=sys.stderr) + return 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/qa/runner/harness_lib.py b/qa/runner/harness_lib.py new file mode 100644 index 0000000..303a7b2 --- /dev/null +++ b/qa/runner/harness_lib.py @@ -0,0 +1,121 @@ +"""Pure configuration, planning, status, and report helpers for Weft QA.""" + +from __future__ import annotations + +import json +from collections import Counter +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Iterable, Mapping, Sequence + +STATUSES = ("PASS", "FAIL", "BLOCKED", "SKIPPED") + + +class HarnessConfigError(ValueError): + """Raised when declarative QA configuration is malformed.""" + + +@dataclass(frozen=True) +class SuitePlan: + name: str + config: Mapping[str, Any] + + +def utc_now() -> datetime: + return datetime.now(timezone.utc) + + +def isoformat_utc(value: datetime) -> str: + return value.astimezone(timezone.utc).isoformat(timespec="milliseconds").replace("+00:00", "Z") + + +def load_config(path: Path) -> dict[str, Any]: + try: + config = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as error: + raise HarnessConfigError(f"cannot load QA config: {error}") from error + if config.get("schemaVersion") != 1 or not isinstance(config.get("profiles"), dict) or not isinstance(config.get("suites"), dict): + raise HarnessConfigError("qa.config.json must contain schemaVersion 1, profiles, and suites") + for name, suite in config["suites"].items(): + if not isinstance(suite, dict) or not isinstance(suite.get("description"), str): + raise HarnessConfigError(f"suite {name!r} must define a description") + command = suite.get("command") + if not isinstance(command, list) or not command or not all(isinstance(item, str) for item in command): + raise HarnessConfigError(f"suite {name!r} must define a non-empty command array") + for field in ("requires", "requiresVariables", "requiredCommands", "requiredAnyCommands", "platforms"): + values = suite.get(field, []) + if not isinstance(values, list) or not all(isinstance(item, str) for item in values): + raise HarnessConfigError(f"suite {name!r} {field} must be a string array") + if suite.get("kind", "deterministic") not in ("deterministic", "observation"): + raise HarnessConfigError(f"suite {name!r} has an invalid kind") + timeout = suite.get("timeoutSeconds", 300) + if not isinstance(timeout, int) or isinstance(timeout, bool) or timeout < 1: + raise HarnessConfigError(f"suite {name!r} timeoutSeconds must be positive") + if "optional" in suite and not isinstance(suite["optional"], bool): + raise HarnessConfigError(f"suite {name!r} optional must be boolean") + codes = suite.get("skipExitCodes", []) + if not isinstance(codes, list) or not all(isinstance(code, int) and not isinstance(code, bool) for code in codes): + raise HarnessConfigError(f"suite {name!r} skipExitCodes must be integers") + for name, suites in config["profiles"].items(): + if not isinstance(suites, list) or not suites or not all(isinstance(item, str) and item in config["suites"] for item in suites): + raise HarnessConfigError(f"profile {name!r} must reference known suites") + return config + + +def resolve_plan(config: Mapping[str, Any], selected: Sequence[str]) -> list[SuitePlan]: + suites = config["suites"] + ordered: list[SuitePlan] = [] + permanent: set[str] = set() + temporary: set[str] = set() + def visit(name: str) -> None: + if name not in suites: + raise HarnessConfigError(f"unknown suite {name!r}") + if name in permanent: + return + if name in temporary: + raise HarnessConfigError(f"suite dependency cycle includes {name!r}") + temporary.add(name) + for dependency in suites[name].get("requires", []): + visit(dependency) + temporary.remove(name) + permanent.add(name) + ordered.append(SuitePlan(name, suites[name])) + for name in selected: + visit(name) + return ordered + + +def render_command(command: Sequence[str], variables: Mapping[str, str]) -> list[str]: + try: + return [item.format_map(variables) for item in command] + except KeyError as error: + raise HarnessConfigError(f"unknown command variable {error.args[0]!r}") from error + + +def summarize_results(results: Iterable[Mapping[str, Any]]) -> dict[str, Any]: + rows = list(results) + counts = Counter(row.get("status") for row in rows) + if any(status not in STATUSES for status in counts): + raise HarnessConfigError("unknown suite status") + verdict = "FAIL" if counts["FAIL"] else "BLOCKED" if counts["BLOCKED"] else "SKIPPED" if rows and counts["SKIPPED"] == len(rows) else "PASS" + return {"pass": counts["PASS"], "fail": counts["FAIL"], "blocked": counts["BLOCKED"], "skipped": counts["SKIPPED"], "total": len(rows), "verdict": verdict} + + +def exit_code_for(summary: Mapping[str, Any]) -> int: + return 1 if summary["fail"] else 2 if summary["blocked"] or summary["verdict"] == "SKIPPED" else 0 + + +def write_report(report: Mapping[str, Any], root: Path) -> tuple[Path, Path]: + directory = root / report["runLabel"] + directory.mkdir(parents=True, exist_ok=True) + latest = directory / "latest.json" + history = directory / f"{report['startedAt'].replace(':', '-').replace('.', '-')}.md" + latest.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + summary = report["summary"] + lines = [f"# Weft QA Report — {report['runLabel']}", "", f"Result: **{summary['verdict']}**", "", "| Suite | Kind | Status | Duration | Note |", "| --- | --- | --- | ---: | --- |"] + for row in report["results"]: + note = str(row.get("note", "")).replace("|", "\\|").replace("\n", "
") + lines.append(f"| {row['name']} | {row.get('kind', 'deterministic')} | {row['status']} | {row.get('durationMs', 0)} ms | {note} |") + history.write_text("\n".join(lines) + "\n", encoding="utf-8") + return latest, history diff --git a/qa/runner/test_harness.py b/qa/runner/test_harness.py new file mode 100644 index 0000000..8833d24 --- /dev/null +++ b/qa/runner/test_harness.py @@ -0,0 +1,54 @@ +import os +import subprocess +import sys +import tempfile +import time +import unittest +from pathlib import Path +from unittest.mock import patch + +import harness +from harness_lib import HarnessConfigError + + +class HarnessLifecycleTests(unittest.TestCase): + def test_required_option_blocks_before_execution(self): + with self.assertRaises(HarnessConfigError): + harness.command_for({"command": ["make"], "requiresVariables": ["version"]}, {"version": ""}) + + @patch("harness.shutil.which", return_value=None) + def test_optional_missing_capability_is_skipped(self, _which): + result = harness._preflight({"optional": True, "requiredCommands": ["but"]}, {}) + self.assertEqual(result, ("SKIPPED", "missing required commands: but")) + + @patch("harness.shutil.which", return_value=None) + def test_missing_alternative_capability_blocks(self, _which): + result = harness._preflight({"requiredAnyCommands": ["docker", "podman"]}, {}) + self.assertEqual(result, ("BLOCKED", "no supported command available: docker, podman")) + + @unittest.skipUnless(os.name == "posix", "process-group cleanup is POSIX-specific") + def test_timeout_terminates_process_group(self): + with tempfile.TemporaryDirectory() as directory: + child_pid = Path(directory) / "child.pid" + command = [ + sys.executable, + "-c", + "import pathlib,subprocess,sys,time; " + "child=subprocess.Popen([sys.executable, '-c', 'import time; time.sleep(60)']); " + f"pathlib.Path({str(child_pid)!r}).write_text(str(child.pid)); time.sleep(60)", + ] + result = harness.run_suite("timeout", {"description": "", "command": command, "timeoutSeconds": 1}, {}) + self.assertEqual((result["status"], result["exitCode"]), ("FAIL", 124)) + pid = int(child_pid.read_text(encoding="utf-8")) + for _ in range(20): + try: + os.kill(pid, 0) + except ProcessLookupError: + break + time.sleep(0.05) + else: + self.fail("timeout left a descendant process running") + + +if __name__ == "__main__": + unittest.main() diff --git a/qa/runner/test_harness_lib.py b/qa/runner/test_harness_lib.py new file mode 100644 index 0000000..63d97ec --- /dev/null +++ b/qa/runner/test_harness_lib.py @@ -0,0 +1,34 @@ +import json +import tempfile +import unittest +from pathlib import Path + +from harness_lib import HarnessConfigError, exit_code_for, load_config, render_command, resolve_plan, summarize_results + + +class HarnessLibraryTests(unittest.TestCase): + def test_config_rejects_invalid_timeout_and_unknown_profile_suite(self): + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "qa.config.json" + path.write_text(json.dumps({"schemaVersion": 1, "profiles": {"smoke": ["missing"]}, "suites": {"suite": {"description": "x", "command": ["true"], "timeoutSeconds": 0}}}), encoding="utf-8") + with self.assertRaises(HarnessConfigError): + load_config(path) + + def test_dependencies_resolve_once_and_cycles_fail(self): + config = {"suites": {"a": {"requires": ["b"]}, "b": {"requires": ["c"]}, "c": {}}} + self.assertEqual([item.name for item in resolve_plan(config, ["a", "b"])], ["c", "b", "a"]) + config["suites"]["c"]["requires"] = ["a"] + with self.assertRaises(HarnessConfigError): + resolve_plan(config, ["a"]) + + def test_rendering_and_status_exit_contract(self): + self.assertEqual(render_command(["make", "VERSION={version}"], {"version": "v0.2.1"}), ["make", "VERSION=v0.2.1"]) + with self.assertRaises(HarnessConfigError): + render_command(["{missing}"], {}) + self.assertEqual(exit_code_for(summarize_results([{"status": "PASS"}, {"status": "SKIPPED"}])), 0) + self.assertEqual(exit_code_for(summarize_results([{"status": "BLOCKED"}])), 2) + self.assertEqual(exit_code_for(summarize_results([{"status": "FAIL"}])), 1) + + +if __name__ == "__main__": + unittest.main()