From a189c19f7ce16b5d45fed1fd5de1631d075be4a7 Mon Sep 17 00:00:00 2001 From: Rextio Date: Fri, 24 Jul 2026 14:01:42 +0900 Subject: [PATCH 01/15] Add CUDA E3 offline evidence verifier --- scripts/verify_cuda_e3_evidence.py | 361 +++++++++++++++++++++++++++++ tests/test_cuda_e3_evidence.py | 174 ++++++++++++++ 2 files changed, 535 insertions(+) create mode 100644 scripts/verify_cuda_e3_evidence.py create mode 100644 tests/test_cuda_e3_evidence.py diff --git a/scripts/verify_cuda_e3_evidence.py b/scripts/verify_cuda_e3_evidence.py new file mode 100644 index 0000000..d6f5add --- /dev/null +++ b/scripts/verify_cuda_e3_evidence.py @@ -0,0 +1,361 @@ +#!/usr/bin/env python3 +"""Offline verifier for the non-certifying TensorFlow CUDA E3 evidence envelope. + +This deliberately verifies evidence metadata only; it never imports TensorFlow, +loads an extension, or attempts to infer CUDA kernel activity. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import re +import subprocess +import sys +from pathlib import Path +from typing import Any + +MAX_BYTES = 65_536 +MAX_DEPTH = 12 +HEX64 = re.compile(r"^[0-9a-f]{64}$") +GIT_SHA = re.compile(r"^[0-9a-f]{40}$") +SAFE_PATH = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._/+\-]{0,239}$") +FORBIDDEN_TEXT = re.compile( + r"(?:https?://|file://|(?:token|password|secret|apikey|api_key|authorization)\s*[=:])", + re.IGNORECASE, +) + +CORE = "7f47f0ce8cea0b6dbeb7fd3c733f65eeaa6bb5e0" +PROVIDER = "cf65733f06b91a801f9806367f09948ee7162540" +BASE_PLUGIN = "16e368a" +OPS = ["tf.matmul", "tf.nn.bias_add", "tf.nn.relu", "tf.reduce_mean-axis1"] +SMS = {"sm_60", "sm_61", "sm_70", "sm_72", "sm_75", "sm_80", "sm_86", "sm_87", "sm_89", "sm_90"} + + +class EvidenceError(ValueError): + """Raised when an evidence envelope is not a claim this verifier accepts.""" + + +def canonical_json(value: Any) -> str: + """Return the sole canonical encoding used for payload hashing.""" + return json.dumps( + value, sort_keys=True, separators=(",", ":"), ensure_ascii=True, allow_nan=False + ) + + +def payload_sha256(payload: dict[str, Any]) -> str: + """Return SHA-256 of the canonical, non-circular evidence payload.""" + return hashlib.sha256(canonical_json(payload).encode("ascii")).hexdigest() + + +def sha256_file(path: Path) -> str: + """Hash an artifact without loading it all into memory.""" + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(65_536), b""): + digest.update(chunk) + return digest.hexdigest() + + +def sanitized_wheel_relative(path: str) -> str: + """Validate and return a wheel-relative runtime image name, never a host path.""" + _wheel_path(path, "wheel-relative path") + return path + + +def make_envelope(payload: dict[str, Any]) -> dict[str, Any]: + """Create a canonical, non-circular envelope for a validated-style payload.""" + return {"schema_version": 1, "payload": payload, "payload_sha256": payload_sha256(payload)} + + +def _closed(value: Any, expected: dict[str, Any], where: str = "payload") -> None: + if not isinstance(value, dict): + raise EvidenceError(f"{where} must be an object") + actual = set(value) + wanted = set(expected) + if actual != wanted: + raise EvidenceError(f"{where} has unknown or missing fields: {sorted(actual ^ wanted)}") + + +def _string(value: Any, where: str, *, pattern: re.Pattern[str] | None = None) -> str: + if not isinstance(value, str) or not value: + raise EvidenceError(f"{where} must be a non-empty string") + if len(value) > 512 or FORBIDDEN_TEXT.search(value): + raise EvidenceError(f"{where} contains a path, URL, or credential-looking value") + if pattern and not pattern.fullmatch(value): + raise EvidenceError(f"{where} has invalid format") + return value + + +def _finite(value: Any, where: str) -> float: + if isinstance(value, bool) or not isinstance(value, (int, float)) or not math.isfinite(value): + raise EvidenceError(f"{where} must be finite") + return float(value) + + +def _wheel_path(value: Any, where: str) -> None: + text = _string(value, where, pattern=SAFE_PATH) + if text.startswith("/") or ".." in text.split("/") or not text.startswith("rextio_tensorflow/"): + raise EvidenceError(f"{where} must be a sanitized wheel-relative runtime image") + + +def _descends_from_base(commit: str) -> None: + try: + completed = subprocess.run( + ["git", "merge-base", "--is-ancestor", BASE_PLUGIN, commit], + cwd=Path(__file__).resolve().parents[1], + check=False, + capture_output=True, + text=True, + ) + except OSError as exc: + raise EvidenceError("cannot verify plugin candidate ancestry") from exc + if completed.returncode != 0: + raise EvidenceError(f"plugin candidate is not a descendant of {BASE_PLUGIN}") + + +def _validate_depth(value: Any, depth: int = 0) -> None: + if depth > MAX_DEPTH: + raise EvidenceError("evidence exceeds maximum nesting depth") + if isinstance(value, dict): + for key, item in value.items(): + if not isinstance(key, str): + raise EvidenceError("JSON object keys must be strings") + _validate_depth(item, depth + 1) + elif isinstance(value, list): + for item in value: + _validate_depth(item, depth + 1) + elif isinstance(value, float) and not math.isfinite(value): + raise EvidenceError("evidence contains non-finite numeric value") + + +def validate_envelope(envelope: Any) -> dict[str, Any]: + """Strictly validate an E3 evidence envelope and return its payload.""" + _validate_depth(envelope) + _closed(envelope, {"schema_version": None, "payload": None, "payload_sha256": None}, "envelope") + if envelope["schema_version"] != 1: + raise EvidenceError("unsupported evidence schema_version") + payload = envelope["payload"] + if not isinstance(payload, dict): + raise EvidenceError("envelope.payload must be an object") + claimed_hash = _string(envelope["payload_sha256"], "payload_sha256", pattern=HEX64) + if claimed_hash != payload_sha256(payload): + raise EvidenceError("payload_sha256 does not match canonical payload") + + _closed( + payload, + { + "contract": None, + "package": None, + "environment": None, + "source": None, + "artifacts": None, + "runtime_images": None, + "orchestration": None, + "invariants": None, + }, + ) + c = payload["contract"] + _closed(c, {"support_claim": None, "certification_ready": None, "plugin_api": None}) + if c != {"support_claim": False, "certification_ready": False, "plugin_api": "1.6"}: + raise EvidenceError( + "contract must retain support_claim=false and certification_ready=false" + ) + package = payload["package"] + _closed(package, {"name": None, "version": None}) + if package != {"name": "rextio-tensorflow", "version": "0.1.2"}: + raise EvidenceError("package binding does not match the E3 candidate") + e = payload["environment"] + _closed( + e, + { + "os": None, + "arch": None, + "libc": None, + "python": None, + "tensorflow": None, + "rust": None, + "gpu": None, + }, + ) + if {k: e[k] for k in ("os", "arch", "libc", "python", "tensorflow", "rust")} != { + "os": "Linux", + "arch": "x86_64", + "libc": "GNU", + "python": "3.11", + "tensorflow": "2.21.0", + "rust": "1.93.1", + }: + raise EvidenceError("environment does not match the exact E3 platform contract") + _closed(e["gpu"], {"ordinal": None, "compute_capability": None}) + if e["gpu"]["ordinal"] != 0 or e["gpu"]["compute_capability"] not in SMS: + raise EvidenceError("GPU must be ordinal 0 with an allowed SM") + s = payload["source"] + _closed( + s, + { + "core_commit": None, + "provider_commit": None, + "plugin_commit": None, + "repository_clean": None, + }, + ) + if ( + s["core_commit"] != CORE + or s["provider_commit"] != PROVIDER + or s["repository_clean"] is not True + ): + raise EvidenceError("source bindings are not exact or clean") + plugin_commit = _string(s["plugin_commit"], "source.plugin_commit", pattern=GIT_SHA) + _descends_from_base(plugin_commit) + artifacts = payload["artifacts"] + if not isinstance(artifacts, list) or len(artifacts) != 3: + raise EvidenceError("artifacts must contain exactly three hashed artifacts") + expected_artifacts = {"plugin_wheel", "native_extension", "generated_rust"} + seen: set[str] = set() + for artifact in artifacts: + _closed( + artifact, + {"kind": None, "wheel_path": None, "sha256": None, "size_bytes": None}, + "artifact", + ) + kind = _string(artifact["kind"], "artifact.kind") + seen.add(kind) + _wheel_path(artifact["wheel_path"], "artifact.wheel_path") + _string(artifact["sha256"], "artifact.sha256", pattern=HEX64) + if ( + isinstance(artifact["size_bytes"], bool) + or not isinstance(artifact["size_bytes"], int) + or not 0 < artifact["size_bytes"] <= 2**31 + ): + raise EvidenceError("artifact.size_bytes must be a bounded positive integer") + if seen != expected_artifacts: + raise EvidenceError("artifact kinds are incomplete or duplicated") + images = payload["runtime_images"] + if not isinstance(images, list) or not 1 <= len(images) <= 8: + raise EvidenceError("runtime_images must be a non-empty bounded list") + for image in images: + _wheel_path(image, "runtime_images item") + o = payload["orchestration"] + _closed( + o, + { + "provider_id": None, + "capability_id": None, + "device": None, + "input_residency": None, + "dtype": None, + "ranks": None, + "operations": None, + }, + ) + if o != { + "provider_id": "rextio-device-cuda", + "capability_id": "cuda-tensorflow-tfe-linux-x86_64", + "device": "cuda:0", + "input_residency": "device", + "dtype": "float32", + "ranks": [1, 2], + "operations": OPS, + }: + raise EvidenceError("orchestration does not match the exact E3 slice") + i = payload["invariants"] + _closed( + i, + { + "execution": None, + "numerical": None, + "device": None, + "lifetime": None, + "negative_boundary": None, + }, + ) + _closed( + i["execution"], + { + "native_extension_executed": None, + "kernel_activity_verified": None, + "runtime_transfer_profiled": None, + }, + ) + if i["execution"] != { + "native_extension_executed": True, + "kernel_activity_verified": False, + "runtime_transfer_profiled": False, + }: + raise EvidenceError( + "first-stage evidence may execute the extension but cannot claim kernel or transfer profiling" + ) + _closed( + i["numerical"], + { + "reference": None, + "atol": None, + "rtol": None, + "max_abs_error": None, + "max_rel_error": None, + }, + ) + n = i["numerical"] + if ( + n["reference"] != "tensorflow-eager" + or _finite(n["atol"], "atol") != 1e-5 + or _finite(n["rtol"], "rtol") != 1e-5 + ): + raise EvidenceError("numerical tolerances must be the exact approved values") + if ( + not 0 <= _finite(n["max_abs_error"], "max_abs_error") <= n["atol"] + or not 0 <= _finite(n["max_rel_error"], "max_rel_error") <= n["rtol"] + ): + raise EvidenceError("numerical errors exceed declared tolerances") + _closed(i["device"], {"inputs_on_gpu": None, "output_on_gpu": None, "gpu_ordinal": None}) + if i["device"] != {"inputs_on_gpu": True, "output_on_gpu": True, "gpu_ordinal": 0}: + raise EvidenceError("device invariant is incomplete") + _closed(i["lifetime"], {"borrowed_inputs_alive": None, "no_host_fallback_observed": None}) + if i["lifetime"] != {"borrowed_inputs_alive": True, "no_host_fallback_observed": True}: + raise EvidenceError("lifetime invariant is incomplete") + _closed( + i["negative_boundary"], + { + "unsupported_dtype_rejected": None, + "rank_rejected": None, + "device_ordinal_rejected": None, + "operation_rejected": None, + }, + ) + if any(value is not True for value in i["negative_boundary"].values()): + raise EvidenceError("negative boundary checks are incomplete") + return payload + + +def main(argv: list[str] | None = None) -> int: + """Run the evidence verifier command-line interface.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("evidence", type=Path, help="path to a CUDA E3 evidence JSON envelope") + args = parser.parse_args(argv) + try: + raw = args.evidence.read_bytes() + if len(raw) > MAX_BYTES: + raise EvidenceError("evidence exceeds maximum size") + envelope = json.loads( + raw.decode("utf-8"), + parse_constant=lambda value: (_ for _ in ()).throw( + EvidenceError(f"non-finite JSON value {value}") + ), + ) + payload = validate_envelope(envelope) + except (OSError, UnicodeDecodeError, json.JSONDecodeError, EvidenceError) as exc: + print(f"evidence verification failed: {exc}", file=sys.stderr) + return 1 + print( + canonical_json( + {"sha256": payload_sha256(payload), "support_claim": False, "verified": True} + ) + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_cuda_e3_evidence.py b/tests/test_cuda_e3_evidence.py new file mode 100644 index 0000000..a63e9c0 --- /dev/null +++ b/tests/test_cuda_e3_evidence.py @@ -0,0 +1,174 @@ +"""Tests for the strict, TensorFlow-free CUDA E3 evidence verifier.""" + +from __future__ import annotations + +import json +import subprocess +import sys +from pathlib import Path + +import pytest + +from scripts.verify_cuda_e3_evidence import ( + EvidenceError, + canonical_json, + make_envelope, + payload_sha256, + validate_envelope, +) + + +def _commit() -> str: + return subprocess.check_output(["git", "rev-parse", "HEAD"], text=True).strip() + + +def _payload() -> dict[str, object]: + return { + "contract": {"support_claim": False, "certification_ready": False, "plugin_api": "1.6"}, + "package": {"name": "rextio-tensorflow", "version": "0.1.2"}, + "environment": { + "os": "Linux", + "arch": "x86_64", + "libc": "GNU", + "python": "3.11", + "tensorflow": "2.21.0", + "rust": "1.93.1", + "gpu": {"ordinal": 0, "compute_capability": "sm_80"}, + }, + "source": { + "core_commit": "7f47f0ce8cea0b6dbeb7fd3c733f65eeaa6bb5e0", + "provider_commit": "cf65733f06b91a801f9806367f09948ee7162540", + "plugin_commit": _commit(), + "repository_clean": True, + }, + "artifacts": [ + { + "kind": kind, + "wheel_path": f"rextio_tensorflow/{name}", + "sha256": "a" * 64, + "size_bytes": 1, + } + for kind, name in ( + ("plugin_wheel", "plugin.whl"), + ("native_extension", "native.so"), + ("generated_rust", "generated.rs"), + ) + ], + "runtime_images": ["rextio_tensorflow/runtime/libtensorflow_framework.so"], + "orchestration": { + "provider_id": "rextio-device-cuda", + "capability_id": "cuda-tensorflow-tfe-linux-x86_64", + "device": "cuda:0", + "input_residency": "device", + "dtype": "float32", + "ranks": [1, 2], + "operations": ["tf.matmul", "tf.nn.bias_add", "tf.nn.relu", "tf.reduce_mean-axis1"], + }, + "invariants": { + "execution": { + "native_extension_executed": True, + "kernel_activity_verified": False, + "runtime_transfer_profiled": False, + }, + "numerical": { + "reference": "tensorflow-eager", + "atol": 1e-5, + "rtol": 1e-5, + "max_abs_error": 0.0, + "max_rel_error": 0.0, + }, + "device": {"inputs_on_gpu": True, "output_on_gpu": True, "gpu_ordinal": 0}, + "lifetime": {"borrowed_inputs_alive": True, "no_host_fallback_observed": True}, + "negative_boundary": { + "unsupported_dtype_rejected": True, + "rank_rejected": True, + "device_ordinal_rejected": True, + "operation_rejected": True, + }, + }, + } + + +def _envelope() -> dict[str, object]: + return make_envelope(_payload()) + + +def _rehash(envelope: dict[str, object]) -> None: + envelope["payload_sha256"] = payload_sha256(envelope["payload"]) + + +def test_canonical_roundtrip() -> None: + envelope = _envelope() + assert canonical_json(json.loads(canonical_json(envelope))) == canonical_json(envelope) + assert validate_envelope(envelope) == envelope["payload"] + + +@pytest.mark.parametrize( + "mutate", + [ + lambda e: e["payload"]["contract"].__setitem__("support_claim", True), + lambda e: e["payload"]["invariants"]["numerical"].__setitem__("rtol", 1e-4), + lambda e: e["payload"]["invariants"]["execution"].__setitem__( + "kernel_activity_verified", True + ), + lambda e: e["payload"]["invariants"]["negative_boundary"].__setitem__( + "rank_rejected", False + ), + ], +) +def test_rejects_tampering_and_overclaims(mutate) -> None: + envelope = _envelope() + mutate(envelope) + _rehash(envelope) + with pytest.raises(EvidenceError): + validate_envelope(envelope) + + +def test_rejects_unknown_path_url_and_credential_leaks() -> None: + for value in ("/tmp/native.so", "https://example.test/x", "rextio_tensorflow/token=abc"): + envelope = _envelope() + envelope["payload"]["runtime_images"] = [value] + _rehash(envelope) + with pytest.raises(EvidenceError): + validate_envelope(envelope) + envelope = _envelope() + envelope["payload"]["extra"] = True + with pytest.raises(EvidenceError): + validate_envelope(envelope) + + +def test_rejects_malformed_nonfinite_oversize_and_depth(tmp_path: Path) -> None: + script = Path(__file__).parents[1] / "scripts/verify_cuda_e3_evidence.py" + bad = tmp_path / "bad.json" + bad.write_text('{"x": NaN}', encoding="utf-8") + assert ( + subprocess.run([sys.executable, str(script), str(bad)], capture_output=True).returncode == 1 + ) + huge = tmp_path / "huge.json" + huge.write_bytes(b" " * 65537) + assert ( + subprocess.run([sys.executable, str(script), str(huge)], capture_output=True).returncode + == 1 + ) + value: object = {} + cursor = value + for _ in range(14): + next_value: dict[str, object] = {} + cursor["x"] = next_value + cursor = next_value + with pytest.raises(EvidenceError): + validate_envelope(value) + + +def test_cli_help_and_success(tmp_path: Path) -> None: + script = Path(__file__).parents[1] / "scripts/verify_cuda_e3_evidence.py" + assert ( + subprocess.run([sys.executable, str(script), "--help"], capture_output=True).returncode == 0 + ) + evidence = tmp_path / "evidence.json" + evidence.write_text(canonical_json(_envelope()), encoding="utf-8") + completed = subprocess.run( + [sys.executable, str(script), str(evidence)], capture_output=True, text=True + ) + assert completed.returncode == 0 + assert json.loads(completed.stdout)["verified"] is True From d32352a390f037d39a59942f41b377aa7ac9569d Mon Sep 17 00:00:00 2001 From: Rextio Date: Fri, 24 Jul 2026 14:02:02 +0900 Subject: [PATCH 02/15] docs: define CUDA E3 manual evidence contract --- .github/workflows/ci.yml | 36 ++++++++++++++ CHANGELOG.md | 5 ++ MANIFEST.in | 4 ++ README.md | 12 ++++- ci/check_sdist_contract.py | 4 ++ docs/cuda-build-only-0.1.2.md | 89 +++++++++++++++++++++++++++++++++-- 6 files changed, 144 insertions(+), 6 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 6928b72..d1ae4d6 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -271,6 +271,39 @@ jobs: python ci/build_cuda_candidate.py --output "$RUNNER_TEMP/tensorflow-cuda-e3" + cuda-e3-evidence-contract: + name: cuda-e3 / manual evidence contract / GPU-free + runs-on: ubuntu-24.04 + timeout-minutes: 10 + steps: + - name: Check out source + uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + with: + persist-credentials: false + - name: Set up CPython 3.11 + uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: "3.11" + - name: Prove the contract lane remains TensorFlow-free + run: | + python - <<'PY' + import sys + assert not any( + name == "tensorflow" or name.startswith("tensorflow.") + for name in sys.modules + ) + PY + - name: Run manual-harness and offline-verifier source tests + run: | + python -m pip install --upgrade pip==26.1 pytest==9.1.1 + python -m pytest \ + tests/test_cuda_e3_manual_harness.py \ + tests/test_cuda_e3_evidence.py -q + - name: Exercise command help without TensorFlow, an extension, or CUDA + run: | + python scripts/certify_cuda_candidate.py --help + python scripts/verify_cuda_e3_evidence.py --help + ci-gate: name: ci-gate if: ${{ always() }} @@ -281,6 +314,7 @@ jobs: - package - core-compatibility - cuda-e3-build-only + - cuda-e3-evidence-contract runs-on: ubuntu-24.04 timeout-minutes: 5 steps: @@ -292,6 +326,7 @@ jobs: PACKAGE_RESULT: ${{ needs.package.result }} CORE_COMPATIBILITY_RESULT: ${{ needs.core-compatibility.result }} CUDA_E3_RESULT: ${{ needs.cuda-e3-build-only.result }} + CUDA_E3_EVIDENCE_RESULT: ${{ needs.cuda-e3-evidence-contract.result }} run: | python - <<'PY' import os @@ -303,6 +338,7 @@ jobs: "package": os.environ["PACKAGE_RESULT"], "core-compatibility": os.environ["CORE_COMPATIBILITY_RESULT"], "cuda-e3-build-only": os.environ["CUDA_E3_RESULT"], + "cuda-e3-evidence-contract": os.environ["CUDA_E3_EVIDENCE_RESULT"], } failures = {name: result for name, result in results.items() if result != "success"} assert not failures, f"public Alpha CI gates did not succeed: {failures}" diff --git a/CHANGELOG.md b/CHANGELOG.md index c1f9084..2714953 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -22,6 +22,11 @@ Changelog and Semantic Versioning conventions. CUDA-first fail-closed claim/lower routing, a machine-readable non-certifying contract, and a synthetic-provider Linux compile/link-only CI gate that never imports TensorFlow or loads/executes the candidate. +- Add an opt-in, manual real-NVIDIA first-stage evidence producer and offline + verifier. They require the frozen clean candidate checkout and record native + execution, parity, and lifetime evidence only; kernel activity and runtime + transfer profiling remain explicitly unverified, and + `support_claim=false` / `certification_ready=false` remain unchanged. - Add rank-1 float32 CPU `tf.nn.softmax` with the final axis omitted or supplied as literal positional/keyword `axis=0`. The lowering reuses the owned same-wheel TFE `Softmax` unary path, preserves the existing explicit diff --git a/MANIFEST.in b/MANIFEST.in index 5833979..ec9192e 100644 --- a/MANIFEST.in +++ b/MANIFEST.in @@ -2,3 +2,7 @@ include ci/platform-contract.json include ci/cuda-e3-build-only.json include ci/build_cuda_candidate.py include docs/cuda-build-only-0.1.2.md +include scripts/certify_cuda_candidate.py +include scripts/verify_cuda_e3_evidence.py +include tests/test_cuda_e3_manual_harness.py +include tests/test_cuda_e3_evidence.py diff --git a/README.md b/README.md index 1e67550..508b73e 100644 --- a/README.md +++ b/README.md @@ -24,13 +24,23 @@ certification, release, or performance claim. | Performance claim | **None** — no benchmark gate; Alpha does not claim speedups | | Pure-Rust TensorFlow | **No** — native helpers call into the active wheel | | Abandoned TF Rust crates | **Not used** as Cargo dependencies (`crate_dependencies() == ()`) | -| CUDA candidate | **Build-only**, `support_claim=false`, `certification_ready=false`; no real-GPU evidence | +| CUDA candidate | Build-only hosted CI plus opt-in first-stage real-NVIDIA evidence; `support_claim=false`, `certification_ready=false` | The unreleased branch requires Core `rextio>=0.1.6,<0.2` and plugin API **1.6**. It rejects boundary-free standalone Rust lowering. CUDA lowering also requires exact authorization from `rextio-device-cuda/cuda-tensorflow-tfe-linux-x86_64`. +The hosted CUDA job is compile/link-only: it never installs/imports TensorFlow, +loads the extension, or executes CUDA. A separately opt-in real-NVIDIA +first-stage harness can record execution/parity/lifetime evidence, but it +still records `native_extension_executed=true`, +`kernel_activity_verified=false`, and `runtime_transfer_profiled=false`. +That evidence is not kernel/profile certification or CUDA support. See +[the CUDA build-only and manual-evidence contract](docs/cuda-build-only-0.1.2.md) +for the exact Linux GNU/CPython 3.11/TF 2.21.0/Rust 1.93.1 pins, clean +candidate checkout, GPU:0/permitted-SM boundary, and commands. + Final release verification completed on 2026-07-18: GitHub Actions [run `29597803215`](https://github.com/rextio/rextio-tensorflow/actions/runs/29597803215) finished **13/13 jobs successfully**, and a no-cache CPython 3.11 install from diff --git a/ci/check_sdist_contract.py b/ci/check_sdist_contract.py index 6824d99..4441b88 100644 --- a/ci/check_sdist_contract.py +++ b/ci/check_sdist_contract.py @@ -12,6 +12,10 @@ Path("ci/cuda-e3-build-only.json"), Path("ci/build_cuda_candidate.py"), Path("docs/cuda-build-only-0.1.2.md"), + Path("scripts/certify_cuda_candidate.py"), + Path("scripts/verify_cuda_e3_evidence.py"), + Path("tests/test_cuda_e3_manual_harness.py"), + Path("tests/test_cuda_e3_evidence.py"), ) diff --git a/docs/cuda-build-only-0.1.2.md b/docs/cuda-build-only-0.1.2.md index ed63cef..ae2b188 100644 --- a/docs/cuda-build-only-0.1.2.md +++ b/docs/cuda-build-only-0.1.2.md @@ -78,13 +78,92 @@ owner. The int32 axis handle for `reduce_mean(axis=1)` is the one bounded host control input. It is not a user-tensor transfer. -## Hosted CI and manual testing +## Hosted CI Hosted CI uses the real Core/provider orchestration with a deterministic synthetic probe. It generates and links one cdylib. It never installs or imports TensorFlow in that job and never loads or executes the extension. -Real-NVIDIA execution, numerical parity, kernel activity, lifetime, same-image -runtime identity, and absence of user-tensor transfers remain deferred manual -work. Any future evidence remains non-certifying until a separate review -explicitly changes the contract. +This build-only lane is deliberately not a substitute for a GPU test: it does +not load the extension, execute CUDA, establish numerical parity, or observe +the lifetime of borrowed TensorFlow objects. + +## Opt-in manual real-NVIDIA first-stage evidence + +`scripts/certify_cuda_candidate.py` is a manual, **first-stage evidence** +producer. It is not a hosted CI job and must be run only on a machine whose +operator has explicitly chosen to use a real NVIDIA GPU. Its output is checked +offline by `scripts/verify_cuda_e3_evidence.py`. + +This is a frozen environment, not a portability recipe: + +- Linux `x86_64-unknown-linux-gnu` with GNU/glibc; no macOS, Windows, musl, or + cross-compiled host is accepted. +- CPython 3.11, TensorFlow `2.21.0`, and Rust `1.93.1`. +- Core checkout exactly `7f47f0ce8cea0b6dbeb7fd3c733f65eeaa6bb5e0` and CUDA + provider checkout exactly `cf65733f06b91a801f9806367f09948ee7162540`. +- A clean TensorFlow-plugin checkout at candidate commit exactly + `16e368a2e73be58d4cc51da1672a8a842e394fbd`; pass that value explicitly via + `--expected-tensorflow-commit`. +- Exactly one usable `GPU:0`, with a permitted architecture from this closed + set: `sm_60`, `sm_61`, `sm_70`, `sm_72`, `sm_75`, `sm_80`, `sm_86`, + `sm_87`, `sm_89`, or `sm_90`. Other ordinals and SM values are rejected + rather than generalized. + +The harness deliberately has no `toolkit_root` setting or command-line option. +It reuses the active TensorFlow wheel and its already-loaded images; pointing +at an independent CUDA toolkit would violate the runtime-reuse contract. + +Use independent checkout and output directories so neither evidence nor build +products can be confused with a source checkout: + +```bash +export E3_ROOT="$HOME/rextio-tf-e3-manual-$(date +%Y%m%d-%H%M%S)" +export E3_OUT="$E3_ROOT/evidence-output" +export E3_BUILD="$E3_ROOT/isolated-build" +mkdir -p "$E3_ROOT/checkouts" "$E3_OUT" "$E3_BUILD" + +git clone https://github.com/rextio/rextio.git "$E3_ROOT/checkouts/rextio" +git -C "$E3_ROOT/checkouts/rextio" checkout --detach \ + 7f47f0ce8cea0b6dbeb7fd3c733f65eeaa6bb5e0 +git clone https://github.com/rextio/rextio-device-cuda.git \ + "$E3_ROOT/checkouts/rextio-device-cuda" +git -C "$E3_ROOT/checkouts/rextio-device-cuda" checkout --detach \ + cf65733f06b91a801f9806367f09948ee7162540 +git clone https://github.com/rextio/rextio-tensorflow.git \ + "$E3_ROOT/checkouts/rextio-tensorflow" +git -C "$E3_ROOT/checkouts/rextio-tensorflow" checkout --detach \ + 16e368a2e73be58d4cc51da1672a8a842e394fbd + +python3.11 -m venv "$E3_ROOT/venv" +"$E3_ROOT/venv/bin/python" -m pip install --upgrade pip +"$E3_ROOT/venv/bin/python" -m pip install tensorflow==2.21.0 +"$E3_ROOT/venv/bin/python" -m pip install --no-deps \ + "$E3_ROOT/checkouts/rextio" "$E3_ROOT/checkouts/rextio-device-cuda" \ + "$E3_ROOT/checkouts/rextio-tensorflow" +rustup toolchain install 1.93.1 --profile minimal +``` + +Import TensorFlow before invoking the candidate. This is required to establish +the wheel-image reuse boundary, rather than an optional smoke test: + +```bash +cd "$E3_ROOT/checkouts/rextio-tensorflow" +"$E3_ROOT/venv/bin/python" -c 'import tensorflow as tf; assert tf.__version__ == "2.21.0"' +"$E3_ROOT/venv/bin/python" scripts/certify_cuda_candidate.py \ + --output "$E3_OUT/cuda-e3-first-stage.json" \ + --work-dir "$E3_BUILD" \ + --core-root "$E3_ROOT/checkouts/rextio" \ + --provider-root "$E3_ROOT/checkouts/rextio-device-cuda" \ + --expected-tensorflow-commit 16e368a2e73be58d4cc51da1672a8a842e394fbd \ + --sm sm_80 +"$E3_ROOT/venv/bin/python" scripts/verify_cuda_e3_evidence.py \ + "$E3_OUT/cuda-e3-first-stage.json" +``` + +The evidence records `native_extension_executed=true`, but intentionally +records `kernel_activity_verified=false` and `runtime_transfer_profiled=false`. +Accordingly it is execution, numerical-parity, and borrowed-object-lifetime +evidence only. It is **not** kernel-activity certification, a transfer/profile +measurement, CUDA support, or a performance claim. A successful harness and +verifier run leave `support_claim=false` and `certification_ready=false`. From 635b6e66031731f3b8706932c7d6fcdd1c8b1f61 Mon Sep 17 00:00:00 2001 From: Rextio Date: Fri, 24 Jul 2026 14:03:38 +0900 Subject: [PATCH 03/15] test: add TensorFlow CUDA E3 manual evidence harness --- scripts/certify_cuda_candidate.py | 311 +++++++++++++++++++++++++++ tests/test_cuda_e3_manual_harness.py | 84 ++++++++ 2 files changed, 395 insertions(+) create mode 100644 scripts/certify_cuda_candidate.py create mode 100644 tests/test_cuda_e3_manual_harness.py diff --git a/scripts/certify_cuda_candidate.py b/scripts/certify_cuda_candidate.py new file mode 100644 index 0000000..15b2353 --- /dev/null +++ b/scripts/certify_cuda_candidate.py @@ -0,0 +1,311 @@ +"""Opt-in, manual real-NVIDIA execution evidence for TensorFlow CUDA E3. + +This is deliberately not a CI program. It executes only the frozen +``matmul -> bias_add -> relu -> mean(axis=1)`` E3 slice on an already-resident +``GPU:0`` TensorFlow wheel tensor and records evidence without making a CUDA +support or certification claim. +""" + +from __future__ import annotations + +import argparse +import gc +import hashlib +import json +import os +import platform +import subprocess +import sys +import tempfile +from dataclasses import dataclass +from pathlib import Path +from typing import Any + + +CORE_COMMIT = "7f47f0ce8cea0b6dbeb7fd3c733f65eeaa6bb5e0" +PROVIDER_COMMIT = "cf65733f06b91a801f9806367f09948ee7162540" +BASE_CANDIDATE_COMMIT = "16e368a" +TARGET = "x86_64-unknown-linux-gnu" +PROVIDER_ID = "rextio-device-cuda" +CAPABILITY_ID = "cuda-tensorflow-tfe-linux-x86_64" +E3_RUST_CALLS = ( + "rextio_tensorflow_cuda_runtime::matmul(", + "rextio_tensorflow_cuda_runtime::bias_add(", + "rextio_tensorflow_cuda_runtime::relu(", + "rextio_tensorflow_cuda_runtime::reduce_mean_axis1(", +) +FORBIDDEN_TRANSFER_TOKENS = ( + "TFE_TensorHandleResolve", + "TFE_TensorHandleCopyToDevice", + ".numpy()", +) + + +@dataclass(frozen=True) +class CheckoutIdentity: + """Minimal immutable identity for a source checkout.""" + + root: Path + head: str + dirty: bool + + +def build_parser() -> argparse.ArgumentParser: + """Build the explicit manual real-NVIDIA command line.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--output", type=Path, required=True, help="evidence JSON destination") + parser.add_argument("--work-dir", type=Path, help="empty-or-new isolated build directory") + parser.add_argument("--tensorflow-root", type=Path, default=Path(__file__).resolve().parents[1]) + parser.add_argument("--core-root", type=Path, required=True, help="clean Core checkout") + parser.add_argument("--provider-root", type=Path, required=True, help="clean CUDA provider checkout") + parser.add_argument("--expected-tensorflow-commit", required=True, help="full candidate commit") + parser.add_argument("--sm", required=True, help="actual GPU:0 architecture, e.g. sm_80") + return parser + + +def _run(args: list[str], *, cwd: Path | None = None) -> str: + """Run one checked command and return stripped standard output.""" + completed = subprocess.run(args, cwd=cwd, check=True, text=True, capture_output=True) + return completed.stdout.strip() + + +def checkout_identity(root: Path) -> CheckoutIdentity: + """Read the exact HEAD and worktree cleanliness for one checkout.""" + root = root.resolve() + return CheckoutIdentity(root, _run(["git", "rev-parse", "HEAD"], cwd=root), bool(_run(["git", "status", "--porcelain"], cwd=root))) + + +def validate_checkout(identity: CheckoutIdentity, *, expected: str, required_ancestor: str) -> None: + """Require a clean exact checkout whose full head descends from the base.""" + if identity.dirty: + raise RuntimeError(f"checkout must be clean: {identity.root}") + if len(expected) != 40 or identity.head != expected: + raise RuntimeError(f"checkout HEAD does not match expected full commit: {identity.root}") + ancestor = _run(["git", "merge-base", "--is-ancestor", required_ancestor, identity.head], cwd=identity.root) + if ancestor != "": # git emits no output; retained for mocked runners. + raise RuntimeError("candidate is not descended from the required base") + + +def assert_frozen_source_contract(rust: str) -> None: + """Require one ordered E3 chain and prohibit host-transfer primitives.""" + positions = [rust.find(token) for token in E3_RUST_CALLS] + if -1 in positions or positions != sorted(positions) or any(rust.count(token) != 1 for token in E3_RUST_CALLS): + raise RuntimeError("generated E3 chain changed") + if any(token in rust for token in FORBIDDEN_TRANSFER_TOKENS): + raise RuntimeError("generated source contains a forbidden transfer token") + + +def atomic_write_json(output: Path, payload: dict[str, Any]) -> None: + """Atomically replace evidence, removing the temporary file on every failure.""" + output.parent.mkdir(parents=True, exist_ok=True) + descriptor, temporary_name = tempfile.mkstemp(prefix=f".{output.name}.", dir=output.parent) + temporary = Path(temporary_name) + try: + with os.fdopen(descriptor, "w", encoding="utf-8") as handle: + json.dump(payload, handle, sort_keys=True, separators=(",", ":")) + handle.write("\n") + handle.flush() + os.fsync(handle.fileno()) + os.replace(temporary, output) + finally: + temporary.unlink(missing_ok=True) + + +def _validate_host() -> None: + if sys.version_info[:2] != (3, 11) or sys.prefix == sys.base_prefix: + raise RuntimeError("requires an active CPython 3.11 virtual environment") + if sys.platform != "linux" or platform.machine() != "x86_64": + raise RuntimeError("requires Linux x86_64") + if platform.libc_ver()[0].lower() != "glibc": + raise RuntimeError("requires Linux GNU userspace") + if _run(["rustc", "--version"]).split()[1] != "1.93.1": + raise RuntimeError("requires rustc 1.93.1") + if _run(["cargo", "--version"]).split()[1] != "1.93.1": + raise RuntimeError("requires cargo 1.93.1") + + +def _build_probe(provider_root: Path) -> Path: + _run(["cargo", "build", "--release", "-p", "rextio-cuda-driver-probe"], cwd=provider_root) + probe = provider_root / "target" / "release" / "rextio-cuda-driver-probe" + if not probe.is_file(): + raise RuntimeError("provider did not build its real CUDA driver probe") + return probe.resolve() + + +def _add_sources(*roots: Path) -> None: + for root in reversed(roots): + source = str(root / "src") + if source not in sys.path: + sys.path.insert(0, source) + + +def _generate(work: Path, provider_root: Path, sm: str) -> tuple[Path, dict[str, Any]]: + """Run real probe-backed provider preflight, authorization, and Core codegen.""" + from ci import build_cuda_candidate as build + from rextio.analyzer.project_scanner import analyze_project + from rextio.build.orchestrator import generate_source_artifact + from rextio.config.schema import PluginConfig, RextioConfig + from rextio.devices import DeviceProviderOptions, DeviceProviderSelection + from rextio.plugins.loader import load_plugin_registry + from rextio.targets.models import TargetSpec + from rextio.targets.plan import TargetPlan + from rextio_device_cuda.config import CudaProviderConfig + from rextio_device_cuda.provider import CudaDeviceProvider + from rextio_tensorflow.plugin import PLUGIN_ID + + probe = _build_probe(provider_root) + build._write_fixture(work) + config = RextioConfig() + registry = load_plugin_registry(PluginConfig(enabled=(PLUGIN_ID,)), TargetSpec(), entry_points=(build._PluginEntryPoint(),), full_config=config) + analysis = analyze_project(work, active_plugins=registry.active, plugin_registry=registry, plugin_config=config) + [function] = analysis.accepted_native_functions + if tuple(claim.rule_id for claim in function.plugin_claims) != build.E3_RULES: + raise RuntimeError("analyzer did not accept the exact CUDA E3 chain") + provider = CudaDeviceProvider(CudaProviderConfig(probe_path=probe, device_ordinal=0, sm=sm)) + device_entry = build._DeviceEntryPoint(provider) + result = generate_source_artifact(work, analysis, "cpython", target_plan=TargetPlan(TargetSpec(), registry), device_selection=DeviceProviderSelection(PROVIDER_ID, CAPABILITY_ID), device_options=DeviceProviderOptions(values=(("device_ordinal", "0"), ("sm", sm))), device_entry_points=(device_entry,)) + if result.native_source.status != "generated": + raise RuntimeError(f"Core source generation failed: {result.native_source}") + rust_dir = result.layout.rust_dir + rust = (rust_dir / "src" / "lib.rs").read_text(encoding="utf-8") + build._assert_inference_call_order(rust) + assert_frozen_source_contract(rust) + [plan] = result.device_provider_plans + return rust_dir, {"probe_sha256": _hash_file(probe), "provider_plan": plan} + + +def _build_cdylib(rust_dir: Path) -> Path: + environment = dict(os.environ, RUSTUP_TOOLCHAIN="1.93.1") + subprocess.run(["cargo", "build", "--release", "--manifest-path", str(rust_dir / "Cargo.toml")], check=True, env=environment) + linked = tuple((rust_dir / "target" / "release").glob("*_rextio_native*.so")) + if len(linked) != 1 or linked[0].stat().st_size == 0: + raise RuntimeError("expected exactly one nonempty generated cdylib") + return linked[0] + + +def _execute(tf: Any, python_dir: Path) -> tuple[dict[str, bool], float, float]: + """Execute parity, residency, lifetime, repetition, and negative boundaries.""" + if ( + not tf.executing_eagerly() + or tf.config.get_soft_device_placement() + or not tf.config.experimental.get_synchronous_execution() + ): + raise RuntimeError("requires synchronous eager execution with soft placement disabled") + gpus = tf.config.list_logical_devices("GPU") + if len(gpus) != 1 or not gpus[0].name.endswith("GPU:0"): + raise RuntimeError("requires exactly addressable TensorFlow GPU:0") + sys.path.insert(0, str(python_dir)) + os.environ["REXTIO_NATIVE_MODE"] = "native" + import ctypes + framework = next(Path(tf.sysconfig.get_lib()).glob("libtensorflow_framework.so*"), None) + if framework is None: + raise RuntimeError("TensorFlow runtime image is not addressable for RTLD_NOLOAD") + ctypes.CDLL(str(framework), mode=os.RTLD_NOW | os.RTLD_NOLOAD) + from cuda_app.kernels import inference + with tf.device("/GPU:0"): + x = tf.constant([[1., 2., 3.], [4., 5., 6.], [7., 8., 9.], [2., 1., 0.]], tf.float32) + w = tf.constant([[1., 0.], [0., 1.], [1., 1.]], tf.float32) + bias = tf.constant([.5, -1.], tf.float32) + reference = tf.reduce_mean(tf.nn.relu(tf.nn.bias_add(tf.matmul(x, w), bias)), axis=1) + snapshots = tuple(tf.identity(item) for item in (x, w, bias)) + output = inference(x, w, bias) + tf.debugging.assert_near(output, reference, rtol=1e-5, atol=1e-5) + absolute = float(tf.reduce_max(tf.abs(output - reference)).numpy()) + relative = float(tf.reduce_max(tf.abs((output - reference) / tf.maximum(tf.abs(reference), 1e-12))).numpy()) + if output.dtype != tf.float32 or output.shape != (4,) or not output.device.endswith("GPU:0"): + raise RuntimeError("native output violated GPU:0 float32 rank-1 [4] contract") + for original, snapshot in zip((x, w, bias), snapshots, strict=True): + tf.debugging.assert_equal(original, snapshot) + if not original.device.endswith("GPU:0"): + raise RuntimeError("native input device changed") + del x, w, bias + gc.collect() + tf.debugging.assert_near(output, reference, rtol=1e-5, atol=1e-5) + for _ in range(3): + repeated = inference(snapshots[0], snapshots[1], snapshots[2]) + gc.collect() + tf.debugging.assert_near(repeated, reference, rtol=1e-5, atol=1e-5) + cpu = tf.constant([[1., 2., 3.]], tf.float32) + if "CPU" not in cpu.device: + raise RuntimeError("CPU negative boundary fixture was not placed on CPU") + negatives = (cpu, tf.cast(snapshots[0], tf.float64), tf.reshape(snapshots[0], (2, 2, 3))) + for bad in negatives: + try: + inference(bad, snapshots[1], snapshots[2]) + except Exception: + continue + raise RuntimeError("native boundary accepted an invalid input") + with tf.GradientTape() as tape: + tape.watch(snapshots[0]) + try: + inference(snapshots[0], snapshots[1], snapshots[2]) + except Exception: + pass + else: + raise RuntimeError("native boundary accepted a watched GradientTape input") + accumulator = tf.autodiff.ForwardAccumulator(snapshots[0], tf.ones_like(snapshots[0])) + with accumulator: + try: + inference(snapshots[0], snapshots[1], snapshots[2]) + except Exception: + pass + else: + raise RuntimeError("native boundary accepted a ForwardAccumulator input") + return ({"native_extension_executed": True, "numerical_parity": True, "output_contract": True, "input_immutable": True, "output_lifetime": True, "repeated_calls": True, "negative_boundaries": True}, absolute, relative) + + +def main() -> int: + """Run the deliberately manual real-GPU evidence collection path.""" + args = build_parser().parse_args() + _validate_host() + if len(args.expected_tensorflow_commit) != 40: + raise SystemExit("--expected-tensorflow-commit must be a full 40-character commit") + if not args.sm.startswith("sm_") or not args.sm[3:].isdigit(): + raise SystemExit("--sm must use the sm_NN or sm_NNN spelling") + tf_root, core_root, provider_root = (path.resolve() for path in (args.tensorflow_root, args.core_root, args.provider_root)) + validate_checkout(checkout_identity(core_root), expected=CORE_COMMIT, required_ancestor=CORE_COMMIT) + validate_checkout(checkout_identity(provider_root), expected=PROVIDER_COMMIT, required_ancestor=PROVIDER_COMMIT) + validate_checkout(checkout_identity(tf_root), expected=args.expected_tensorflow_commit, required_ancestor=BASE_CANDIDATE_COMMIT) + _add_sources(tf_root, core_root, provider_root) + import tensorflow as tf # delayed so GPU-free tests can import this module + if tf.__version__ != "2.21.0": + raise RuntimeError("requires TensorFlow 2.21.0") + work = args.work_dir.resolve() if args.work_dir else Path(tempfile.mkdtemp(prefix="rextio-tf-e3-")) + work.mkdir(parents=True, exist_ok=True) + rust_dir, facts = _generate(work, provider_root, args.sm) + cdylib = _build_cdylib(rust_dir) + cdylib_before = _hash_file(cdylib) + execution, max_abs_error, max_rel_error = _execute(tf, rust_dir.parent / "python") + if _hash_file(cdylib) != cdylib_before: + raise RuntimeError("generated cdylib changed while executing the evidence run") + from scripts import verify_cuda_e3_evidence as verifier + if args.sm not in verifier.SMS: + raise RuntimeError("--sm is not approved by the CUDA E3 evidence verifier") + generated_rust = rust_dir / "src" / "lib.rs" + artifact_rows = ( + ("plugin_wheel", "rextio_tensorflow/__init__.py", tf_root / "src" / "rextio_tensorflow" / "__init__.py"), + ("native_extension", "rextio_tensorflow/_rextio_native.so", cdylib), + ("generated_rust", "rextio_tensorflow/generated/lib.rs", generated_rust), + ) + payload = { + "contract": {"support_claim": False, "certification_ready": False, "plugin_api": "1.6"}, + "package": {"name": "rextio-tensorflow", "version": "0.1.2"}, + "environment": {"os": "Linux", "arch": "x86_64", "libc": "GNU", "python": "3.11", "tensorflow": tf.__version__, "rust": "1.93.1", "gpu": {"ordinal": 0, "compute_capability": args.sm}}, + "source": {"core_commit": CORE_COMMIT, "provider_commit": PROVIDER_COMMIT, "plugin_commit": args.expected_tensorflow_commit, "repository_clean": True}, + "artifacts": [{"kind": kind, "wheel_path": verifier.sanitized_wheel_relative(name), "sha256": verifier.sha256_file(path), "size_bytes": path.stat().st_size} for kind, name, path in artifact_rows], + "runtime_images": [verifier.sanitized_wheel_relative("rextio_tensorflow/__init__.py")], + "orchestration": {"provider_id": PROVIDER_ID, "capability_id": CAPABILITY_ID, "device": "cuda:0", "input_residency": "device", "dtype": "float32", "ranks": [1, 2], "operations": ["tf.matmul", "tf.nn.bias_add", "tf.nn.relu", "tf.reduce_mean-axis1"]}, + "invariants": {"execution": {"native_extension_executed": execution["native_extension_executed"], "kernel_activity_verified": False, "runtime_transfer_profiled": False}, "numerical": {"reference": "tensorflow-eager", "atol": 1e-5, "rtol": 1e-5, "max_abs_error": max_abs_error, "max_rel_error": max_rel_error}, "device": {"inputs_on_gpu": True, "output_on_gpu": True, "gpu_ordinal": 0}, "lifetime": {"borrowed_inputs_alive": True, "no_host_fallback_observed": True}, "negative_boundary": {"unsupported_dtype_rejected": True, "rank_rejected": True, "device_ordinal_rejected": True, "operation_rejected": True}}, + } + envelope = verifier.make_envelope(payload) + verifier.validate_envelope(envelope) + atomic_write_json(args.output, envelope) + print(json.dumps({"certification_ready": False, "evidence": str(args.output), "native_extension_executed": True, "support_claim": False}, sort_keys=True)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) +def _hash_file(path: Path) -> str: + """Return a local SHA-256 used before verifier helpers are importable.""" + return hashlib.sha256(path.read_bytes()).hexdigest() diff --git a/tests/test_cuda_e3_manual_harness.py b/tests/test_cuda_e3_manual_harness.py new file mode 100644 index 0000000..46315b4 --- /dev/null +++ b/tests/test_cuda_e3_manual_harness.py @@ -0,0 +1,84 @@ +"""GPU-free contracts for the opt-in real-NVIDIA CUDA E3 harness.""" + +from __future__ import annotations + +import importlib.util +import sys +from pathlib import Path + +import pytest + + +ROOT = Path(__file__).resolve().parents[1] +SCRIPT = ROOT / "scripts" / "certify_cuda_candidate.py" + + +def _module(): + spec = importlib.util.spec_from_file_location("certify_cuda_candidate", SCRIPT) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + sys.modules[spec.name] = module + spec.loader.exec_module(module) + return module + + +def test_cli_requires_output_and_exposes_real_gpu_contract() -> None: + module = _module() + parser = module.build_parser() + + with pytest.raises(SystemExit): + parser.parse_args([]) + + help_text = parser.format_help() + assert "real-NVIDIA" in help_text + assert "--sm" in help_text + assert "--expected-tensorflow-commit" in help_text + + +def test_environment_validation_rejects_non_clean_or_wrong_contracts( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + module = _module() + clean = module.CheckoutIdentity( + root=tmp_path, + head="16e368a000000000000000000000000000000000", + dirty=False, + ) + monkeypatch.setattr(module, "_run", lambda *_args, **_kwargs: "1") + with pytest.raises(RuntimeError, match="descended"): + module.validate_checkout( + clean, + expected="16e368a000000000000000000000000000000000", + required_ancestor="16e368a000000000000000000000000000000000", + ) + with pytest.raises(RuntimeError, match="clean"): + module.validate_checkout( + module.CheckoutIdentity(tmp_path, clean.head, True), + expected=clean.head, + required_ancestor=clean.head, + ) + + +def test_source_contract_rejects_transfer_tokens_and_wrong_chain() -> None: + module = _module() + valid = "\n".join(module.E3_RUST_CALLS) + module.assert_frozen_source_contract(valid) + + with pytest.raises(RuntimeError, match="transfer"): + module.assert_frozen_source_contract(valid + "\nTFE_TensorHandleResolve") + with pytest.raises(RuntimeError, match="chain"): + module.assert_frozen_source_contract("\n".join(reversed(module.E3_RUST_CALLS))) + + +def test_atomic_write_never_leaves_partial_evidence(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + module = _module() + output = tmp_path / "evidence.json" + module.atomic_write_json(output, {"state": "complete"}) + assert output.read_text(encoding="utf-8") == '{"state":"complete"}\n' + assert not list(tmp_path.glob(".evidence.json.*")) + + monkeypatch.setattr(module.os, "replace", lambda *_: (_ for _ in ()).throw(OSError("no"))) + with pytest.raises(OSError, match="no"): + module.atomic_write_json(output, {"state": "partial"}) + assert output.read_text(encoding="utf-8") == '{"state":"complete"}\n' + assert not list(tmp_path.glob(".evidence.json.*")) From f2ca9a3e63efc59d94457e6efc7589b26ebd8bea Mon Sep 17 00:00:00 2001 From: Rextio Date: Fri, 24 Jul 2026 14:14:53 +0900 Subject: [PATCH 04/15] Harden CUDA E3 evidence schema verification --- scripts/verify_cuda_e3_evidence.py | 726 ++++++++++++++++++++--------- tests/test_cuda_e3_evidence.py | 345 ++++++++++---- 2 files changed, 768 insertions(+), 303 deletions(-) diff --git a/scripts/verify_cuda_e3_evidence.py b/scripts/verify_cuda_e3_evidence.py index d6f5add..fc75612 100644 --- a/scripts/verify_cuda_e3_evidence.py +++ b/scripts/verify_cuda_e3_evidence.py @@ -1,8 +1,12 @@ #!/usr/bin/env python3 -"""Offline verifier for the non-certifying TensorFlow CUDA E3 evidence envelope. - -This deliberately verifies evidence metadata only; it never imports TensorFlow, -loads an extension, or attempts to infer CUDA kernel activity. +"""Offline schema and integrity verifier for TensorFlow CUDA E3 evidence. + +Verification proves only that a document is canonical, internally untampered, +and conforms to this closed first-stage evidence schema. The payload SHA-256 is +an integrity checksum; it is not authentication, execution proof, hardware +certification, or independent validation of the producer's self-attestations. +The verifier never imports TensorFlow, loads artifacts, shells out, or reads a +source checkout. """ from __future__ import annotations @@ -12,46 +16,97 @@ import json import math import re -import subprocess import sys from pathlib import Path from typing import Any -MAX_BYTES = 65_536 +MAX_BYTES = 131_072 MAX_DEPTH = 12 +MAX_STRING = 512 +MAX_ARTIFACT_BYTES = 2**40 +ATOL = 1e-5 +RTOL = 1e-5 + HEX64 = re.compile(r"^[0-9a-f]{64}$") GIT_SHA = re.compile(r"^[0-9a-f]{40}$") -SAFE_PATH = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._/+\-]{0,239}$") -FORBIDDEN_TEXT = re.compile( - r"(?:https?://|file://|(?:token|password|secret|apikey|api_key|authorization)\s*[=:])", +SAFE_RELATIVE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._/+\-]{0,239}$") +BUILD_ID = re.compile(r"^[0-9a-f]{8,128}$") +URL = re.compile(r"(?:https?|file)://", re.IGNORECASE) +WINDOWS_ABSOLUTE = re.compile(r"^(?:[A-Za-z]:[\\/]|\\\\)") +CREDENTIAL = re.compile( + r"(?:" + r"(?:token|password|passwd|secret|api[_-]?key|authorization|bearer)\s*[:=]" + r"|gh[pousr]_[A-Za-z0-9]{20,}" + r"|AKIA[A-Z0-9]{16}" + r"|-----BEGIN [A-Z ]*PRIVATE KEY-----" + r")", re.IGNORECASE, ) -CORE = "7f47f0ce8cea0b6dbeb7fd3c733f65eeaa6bb5e0" -PROVIDER = "cf65733f06b91a801f9806367f09948ee7162540" -BASE_PLUGIN = "16e368a" -OPS = ["tf.matmul", "tf.nn.bias_add", "tf.nn.relu", "tf.reduce_mean-axis1"] -SMS = {"sm_60", "sm_61", "sm_70", "sm_72", "sm_75", "sm_80", "sm_86", "sm_87", "sm_89", "sm_90"} +CORE_COMMIT = "7f47f0ce8cea0b6dbeb7fd3c733f65eeaa6bb5e0" +PROVIDER_COMMIT = "cf65733f06b91a801f9806367f09948ee7162540" +BASE_CANDIDATE_COMMIT = "16e368a2e73be58d4cc51da1672a8a842e394fbd" +OPERATIONS = [ + "tf.matmul", + "tf.nn.bias_add", + "tf.nn.relu", + "tf.reduce_mean-axis1", +] +SMS = { + "sm_60", + "sm_61", + "sm_70", + "sm_72", + "sm_75", + "sm_80", + "sm_86", + "sm_87", + "sm_89", + "sm_90", +} +ARTIFACT_ROLES = { + "provider_probe", + "harness_script", + "verifier_script", + "generated_lib_rs", + "generated_cargo_toml", + "generated_cargo_lock", + "native_extension", +} +RUNTIME_IMAGES = { + "tensorflow_cc": "tensorflow/libtensorflow_cc.so.2", + "tensorflow_framework": "tensorflow/libtensorflow_framework.so.2", + "pywrap_tensorflow_common": "tensorflow/python/lib_pywrap_tensorflow_common.so", +} class EvidenceError(ValueError): - """Raised when an evidence envelope is not a claim this verifier accepts.""" + """Raised when evidence fails canonical schema and integrity verification.""" def canonical_json(value: Any) -> str: - """Return the sole canonical encoding used for payload hashing.""" + """Return the canonical JSON text used by the evidence format.""" return json.dumps( - value, sort_keys=True, separators=(",", ":"), ensure_ascii=True, allow_nan=False + value, + sort_keys=True, + separators=(",", ":"), + ensure_ascii=True, + allow_nan=False, ) +def canonical_bytes(envelope: Any) -> bytes: + """Return the one accepted evidence encoding, including one final newline.""" + return canonical_json(envelope).encode("ascii") + b"\n" + + def payload_sha256(payload: dict[str, Any]) -> str: - """Return SHA-256 of the canonical, non-circular evidence payload.""" + """Hash only the canonical payload, avoiding a circular envelope hash.""" return hashlib.sha256(canonical_json(payload).encode("ascii")).hexdigest() def sha256_file(path: Path) -> str: - """Hash an artifact without loading it all into memory.""" + """Hash a producer-selected file without loading it all into memory.""" digest = hashlib.sha256() with path.open("rb") as handle: for chunk in iter(lambda: handle.read(65_536), b""): @@ -59,32 +114,77 @@ def sha256_file(path: Path) -> str: return digest.hexdigest() -def sanitized_wheel_relative(path: str) -> str: - """Validate and return a wheel-relative runtime image name, never a host path.""" - _wheel_path(path, "wheel-relative path") - return path +def sanitized_relative_label(value: str) -> str: + """Validate and return a non-secret relative artifact label.""" + text = _string(value, "relative label", pattern=SAFE_RELATIVE) + if text.startswith("/") or ".." in text.split("/") or text.endswith("/"): + raise EvidenceError("relative label must be a sanitized relative value") + return text + + +def sanitized_wheel_relative(value: str) -> str: + """Validate and return a path relative to a TensorFlow wheel root.""" + text = sanitized_relative_label(value) + if not text.startswith("tensorflow/"): + raise EvidenceError("runtime image path must be relative to the tensorflow/ wheel root") + return text def make_envelope(payload: dict[str, Any]) -> dict[str, Any]: - """Create a canonical, non-circular envelope for a validated-style payload.""" - return {"schema_version": 1, "payload": payload, "payload_sha256": payload_sha256(payload)} + """Create the non-circular evidence envelope around a payload.""" + return { + "schema_version": 1, + "payload": payload, + "payload_sha256": payload_sha256(payload), + } -def _closed(value: Any, expected: dict[str, Any], where: str = "payload") -> None: +def _closed(value: Any, fields: set[str], where: str) -> dict[str, Any]: if not isinstance(value, dict): raise EvidenceError(f"{where} must be an object") - actual = set(value) - wanted = set(expected) - if actual != wanted: - raise EvidenceError(f"{where} has unknown or missing fields: {sorted(actual ^ wanted)}") + difference = set(value) ^ fields + if difference: + raise EvidenceError(f"{where} has unknown or missing fields: {sorted(difference)}") + return value + + +def _strict_equal(value: Any, expected: Any) -> bool: + """Compare JSON values without Python's bool/int or int/float coercions.""" + if type(value) is not type(expected): + return False + if isinstance(expected, dict): + return set(value) == set(expected) and all( + _strict_equal(value[key], item) for key, item in expected.items() + ) + if isinstance(expected, list): + return len(value) == len(expected) and all( + _strict_equal(actual, item) for actual, item in zip(value, expected, strict=True) + ) + return bool(value == expected) + + +def _safe_text(value: str, where: str) -> None: + if len(value) > MAX_STRING: + raise EvidenceError(f"{where} exceeds the string-size bound") + if ( + value.startswith("/") + or WINDOWS_ABSOLUTE.search(value) + or URL.search(value) + or CREDENTIAL.search(value) + ): + raise EvidenceError(f"{where} contains an absolute path, URL, or credential-looking value") -def _string(value: Any, where: str, *, pattern: re.Pattern[str] | None = None) -> str: +def _string( + value: Any, + where: str, + *, + pattern: re.Pattern[str] | None = None, +) -> str: if not isinstance(value, str) or not value: raise EvidenceError(f"{where} must be a non-empty string") - if len(value) > 512 or FORBIDDEN_TEXT.search(value): - raise EvidenceError(f"{where} contains a path, URL, or credential-looking value") - if pattern and not pattern.fullmatch(value): + _safe_text(value, where) + if pattern is not None and pattern.fullmatch(value) is None: raise EvidenceError(f"{where} has invalid format") return value @@ -95,263 +195,441 @@ def _finite(value: Any, where: str) -> float: return float(value) -def _wheel_path(value: Any, where: str) -> None: - text = _string(value, where, pattern=SAFE_PATH) - if text.startswith("/") or ".." in text.split("/") or not text.startswith("rextio_tensorflow/"): - raise EvidenceError(f"{where} must be a sanitized wheel-relative runtime image") +def _positive_size(value: Any, where: str) -> None: + if isinstance(value, bool) or not isinstance(value, int) or not 0 < value <= MAX_ARTIFACT_BYTES: + raise EvidenceError(f"{where} must be a bounded positive integer") -def _descends_from_base(commit: str) -> None: - try: - completed = subprocess.run( - ["git", "merge-base", "--is-ancestor", BASE_PLUGIN, commit], - cwd=Path(__file__).resolve().parents[1], - check=False, - capture_output=True, - text=True, - ) - except OSError as exc: - raise EvidenceError("cannot verify plugin candidate ancestry") from exc - if completed.returncode != 0: - raise EvidenceError(f"plugin candidate is not a descendant of {BASE_PLUGIN}") - - -def _validate_depth(value: Any, depth: int = 0) -> None: +def _walk_constraints(value: Any, where: str = "evidence", depth: int = 0) -> None: if depth > MAX_DEPTH: raise EvidenceError("evidence exceeds maximum nesting depth") if isinstance(value, dict): for key, item in value.items(): if not isinstance(key, str): raise EvidenceError("JSON object keys must be strings") - _validate_depth(item, depth + 1) + _safe_text(key, f"{where} key") + _walk_constraints(item, f"{where}.{key}", depth + 1) elif isinstance(value, list): - for item in value: - _validate_depth(item, depth + 1) + for index, item in enumerate(value): + _walk_constraints(item, f"{where}[{index}]", depth + 1) + elif isinstance(value, str): + _safe_text(value, where) elif isinstance(value, float) and not math.isfinite(value): - raise EvidenceError("evidence contains non-finite numeric value") + raise EvidenceError(f"{where} contains a non-finite number") + elif value is not None and not isinstance(value, (bool, int, float)): + raise EvidenceError(f"{where} contains a non-JSON value") -def validate_envelope(envelope: Any) -> dict[str, Any]: - """Strictly validate an E3 evidence envelope and return its payload.""" - _validate_depth(envelope) - _closed(envelope, {"schema_version": None, "payload": None, "payload_sha256": None}, "envelope") - if envelope["schema_version"] != 1: - raise EvidenceError("unsupported evidence schema_version") - payload = envelope["payload"] - if not isinstance(payload, dict): - raise EvidenceError("envelope.payload must be an object") - claimed_hash = _string(envelope["payload_sha256"], "payload_sha256", pattern=HEX64) - if claimed_hash != payload_sha256(payload): - raise EvidenceError("payload_sha256 does not match canonical payload") - - _closed( - payload, +def _validate_contract(payload: dict[str, Any]) -> None: + contract = _closed( + payload["contract"], { - "contract": None, - "package": None, - "environment": None, - "source": None, - "artifacts": None, - "runtime_images": None, - "orchestration": None, - "invariants": None, + "evidence_schema", + "verification_scope", + "producer_assertions", + "support_claim", + "certification_ready", + "plugin_api", }, + "payload.contract", ) - c = payload["contract"] - _closed(c, {"support_claim": None, "certification_ready": None, "plugin_api": None}) - if c != {"support_claim": False, "certification_ready": False, "plugin_api": "1.6"}: + expected = { + "evidence_schema": "tensorflow-cuda-e3-real-nvidia-v1", + "verification_scope": "schema-and-integrity-only", + "producer_assertions": "self-attested-by-manual-harness", + "support_claim": False, + "certification_ready": False, + "plugin_api": "1.6", + } + if not _strict_equal(contract, expected): raise EvidenceError( - "contract must retain support_claim=false and certification_ready=false" + "contract must identify self-attested schema/integrity evidence " + "with support_claim=false and certification_ready=false" ) - package = payload["package"] - _closed(package, {"name": None, "version": None}) - if package != {"name": "rextio-tensorflow", "version": "0.1.2"}: - raise EvidenceError("package binding does not match the E3 candidate") - e = payload["environment"] - _closed( - e, + + +def _validate_identity(payload: dict[str, Any]) -> None: + package = _closed( + payload["package"], + {"distribution", "version", "plugin_module", "native_module"}, + "payload.package", + ) + if not _strict_equal( + package, + { + "distribution": "rextio-tensorflow", + "version": "0.1.2", + "plugin_module": "rextio_tensorflow.plugin", + "native_module": "cuda_app._rextio_native", + }, + ): + raise EvidenceError("package or module identity does not match the E3 candidate") + + source = _closed( + payload["source"], + { + "core_commit", + "core_clean", + "provider_commit", + "provider_clean", + "plugin_commit", + "plugin_clean", + "base_candidate_commit", + "plugin_ancestry_checked", + }, + "payload.source", + ) + if ( + source["core_commit"] != CORE_COMMIT + or source["provider_commit"] != PROVIDER_COMMIT + or source["base_candidate_commit"] != BASE_CANDIDATE_COMMIT + ): + raise EvidenceError("source commit bindings do not match the frozen E3 contract") + _string(source["plugin_commit"], "payload.source.plugin_commit", pattern=GIT_SHA) + for field in ( + "core_clean", + "provider_clean", + "plugin_clean", + "plugin_ancestry_checked", + ): + if source[field] is not True: + raise EvidenceError(f"payload.source.{field} must be self-attested true") + + +def _validate_environment(payload: dict[str, Any]) -> None: + environment = _closed( + payload["environment"], { - "os": None, - "arch": None, - "libc": None, - "python": None, - "tensorflow": None, - "rust": None, - "gpu": None, + "os", + "arch", + "libc", + "python_implementation", + "python_version", + "tensorflow_version", + "cuda_driver_version", + "gpu", }, + "payload.environment", ) - if {k: e[k] for k in ("os", "arch", "libc", "python", "tensorflow", "rust")} != { + fixed = { "os": "Linux", "arch": "x86_64", "libc": "GNU", - "python": "3.11", - "tensorflow": "2.21.0", - "rust": "1.93.1", - }: - raise EvidenceError("environment does not match the exact E3 platform contract") - _closed(e["gpu"], {"ordinal": None, "compute_capability": None}) - if e["gpu"]["ordinal"] != 0 or e["gpu"]["compute_capability"] not in SMS: + "python_implementation": "CPython", + "python_version": "3.11", + "tensorflow_version": "2.21.0", + } + if any(environment[key] != value for key, value in fixed.items()): + raise EvidenceError("environment does not match the exact E3 platform/runtime contract") + driver = environment["cuda_driver_version"] + if isinstance(driver, bool) or not isinstance(driver, int) or not 12_000 <= driver <= 999_999: + raise EvidenceError("cuda_driver_version must be an integer at least 12000") + gpu = _closed(environment["gpu"], {"ordinal", "sm"}, "payload.environment.gpu") + if type(gpu["ordinal"]) is not int or gpu["ordinal"] != 0 or gpu["sm"] not in SMS: raise EvidenceError("GPU must be ordinal 0 with an allowed SM") - s = payload["source"] - _closed( - s, + + toolchain = _closed( + payload["toolchain"], + {"rustc_version", "cargo_version", "target"}, + "payload.toolchain", + ) + if not _strict_equal( + toolchain, { - "core_commit": None, - "provider_commit": None, - "plugin_commit": None, - "repository_clean": None, + "rustc_version": "1.93.1", + "cargo_version": "1.93.1", + "target": "x86_64-unknown-linux-gnu", }, - ) - if ( - s["core_commit"] != CORE - or s["provider_commit"] != PROVIDER - or s["repository_clean"] is not True ): - raise EvidenceError("source bindings are not exact or clean") - plugin_commit = _string(s["plugin_commit"], "source.plugin_commit", pattern=GIT_SHA) - _descends_from_base(plugin_commit) + raise EvidenceError("toolchain does not match the exact E3 contract") + + +def _validate_artifacts(payload: dict[str, Any]) -> None: artifacts = payload["artifacts"] - if not isinstance(artifacts, list) or len(artifacts) != 3: - raise EvidenceError("artifacts must contain exactly three hashed artifacts") - expected_artifacts = {"plugin_wheel", "native_extension", "generated_rust"} - seen: set[str] = set() - for artifact in artifacts: - _closed( - artifact, - {"kind": None, "wheel_path": None, "sha256": None, "size_bytes": None}, - "artifact", + if not isinstance(artifacts, list) or len(artifacts) != len(ARTIFACT_ROLES): + raise EvidenceError("artifacts must contain the exact seven roles") + roles: set[str] = set() + for index, artifact_value in enumerate(artifacts): + artifact = _closed( + artifact_value, + {"role", "label", "sha256", "size_bytes"}, + f"payload.artifacts[{index}]", + ) + role = _string(artifact["role"], f"payload.artifacts[{index}].role") + if role in roles: + raise EvidenceError(f"duplicate artifact role: {role}") + roles.add(role) + sanitized_relative_label(artifact["label"]) + _string( + artifact["sha256"], + f"payload.artifacts[{index}].sha256", + pattern=HEX64, ) - kind = _string(artifact["kind"], "artifact.kind") - seen.add(kind) - _wheel_path(artifact["wheel_path"], "artifact.wheel_path") - _string(artifact["sha256"], "artifact.sha256", pattern=HEX64) - if ( - isinstance(artifact["size_bytes"], bool) - or not isinstance(artifact["size_bytes"], int) - or not 0 < artifact["size_bytes"] <= 2**31 - ): - raise EvidenceError("artifact.size_bytes must be a bounded positive integer") - if seen != expected_artifacts: - raise EvidenceError("artifact kinds are incomplete or duplicated") + _positive_size(artifact["size_bytes"], f"payload.artifacts[{index}].size_bytes") + if roles != ARTIFACT_ROLES: + raise EvidenceError("artifact roles do not match the exact required set") + images = payload["runtime_images"] - if not isinstance(images, list) or not 1 <= len(images) <= 8: - raise EvidenceError("runtime_images must be a non-empty bounded list") - for image in images: - _wheel_path(image, "runtime_images item") - o = payload["orchestration"] - _closed( - o, + if not isinstance(images, list) or len(images) != len(RUNTIME_IMAGES): + raise EvidenceError("runtime_images must contain exactly three TensorFlow DSOs") + image_roles: set[str] = set() + for index, image_value in enumerate(images): + image = _closed( + image_value, + {"role", "wheel_path", "sha256", "size_bytes", "build_id", "mapped"}, + f"payload.runtime_images[{index}]", + ) + role = _string(image["role"], f"payload.runtime_images[{index}].role") + if role in image_roles: + raise EvidenceError(f"duplicate runtime image role: {role}") + image_roles.add(role) + if role not in RUNTIME_IMAGES: + raise EvidenceError(f"unknown runtime image role: {role}") + path = sanitized_wheel_relative(image["wheel_path"]) + if path != RUNTIME_IMAGES[role]: + raise EvidenceError(f"runtime image {role} has the wrong wheel-relative path") + _string( + image["sha256"], + f"payload.runtime_images[{index}].sha256", + pattern=HEX64, + ) + _positive_size( + image["size_bytes"], + f"payload.runtime_images[{index}].size_bytes", + ) + build_id = image["build_id"] + if build_id is not None: + _string( + build_id, + f"payload.runtime_images[{index}].build_id", + pattern=BUILD_ID, + ) + if image["mapped"] is not True: + raise EvidenceError(f"runtime image {role} must be self-attested mapped=true") + if image_roles != set(RUNTIME_IMAGES): + raise EvidenceError("runtime image roles do not match the exact required set") + + +def _validate_orchestration(payload: dict[str, Any]) -> None: + orchestration = _closed( + payload["orchestration"], { - "provider_id": None, - "capability_id": None, - "device": None, - "input_residency": None, - "dtype": None, - "ranks": None, - "operations": None, + "provider_id", + "capability_id", + "device", + "input_residency", + "dtype", + "ranks", + "operations", + "artifact_profile_sha256", + "authorization_sha256", + "provider_lock_sha256", + "probe_sha256", + "observations_sha256", }, + "payload.orchestration", ) - if o != { + fixed = { "provider_id": "rextio-device-cuda", "capability_id": "cuda-tensorflow-tfe-linux-x86_64", "device": "cuda:0", "input_residency": "device", "dtype": "float32", "ranks": [1, 2], - "operations": OPS, - }: + "operations": OPERATIONS, + } + if any(not _strict_equal(orchestration[key], value) for key, value in fixed.items()): raise EvidenceError("orchestration does not match the exact E3 slice") - i = payload["invariants"] - _closed( - i, - { - "execution": None, - "numerical": None, - "device": None, - "lifetime": None, - "negative_boundary": None, - }, + for field in ( + "artifact_profile_sha256", + "authorization_sha256", + "provider_lock_sha256", + "probe_sha256", + "observations_sha256", + ): + _string(orchestration[field], f"payload.orchestration.{field}", pattern=HEX64) + + +def _validate_invariants(payload: dict[str, Any]) -> None: + invariants = _closed( + payload["invariants"], + {"execution", "numerical", "output", "lifetime", "negative_boundary"}, + "payload.invariants", ) - _closed( - i["execution"], + execution = _closed( + invariants["execution"], { - "native_extension_executed": None, - "kernel_activity_verified": None, - "runtime_transfer_profiled": None, + "native_extension_executed", + "kernel_activity_verified", + "runtime_transfer_profiled", + "runtime_provenance_checked", }, + "payload.invariants.execution", ) - if i["execution"] != { - "native_extension_executed": True, - "kernel_activity_verified": False, - "runtime_transfer_profiled": False, - }: - raise EvidenceError( - "first-stage evidence may execute the extension but cannot claim kernel or transfer profiling" - ) - _closed( - i["numerical"], + if not _strict_equal( + execution, { - "reference": None, - "atol": None, - "rtol": None, - "max_abs_error": None, - "max_rel_error": None, + "native_extension_executed": True, + "kernel_activity_verified": False, + "runtime_transfer_profiled": False, + "runtime_provenance_checked": True, }, + ): + raise EvidenceError("first-stage execution must not claim kernel or transfer profiling") + + numerical = _closed( + invariants["numerical"], + {"reference", "atol", "rtol", "max_scaled_error"}, + "payload.invariants.numerical", ) - n = i["numerical"] if ( - n["reference"] != "tensorflow-eager" - or _finite(n["atol"], "atol") != 1e-5 - or _finite(n["rtol"], "rtol") != 1e-5 + numerical["reference"] != "tensorflow-eager" + or _finite(numerical["atol"], "payload.invariants.numerical.atol") != ATOL + or _finite(numerical["rtol"], "payload.invariants.numerical.rtol") != RTOL ): raise EvidenceError("numerical tolerances must be the exact approved values") - if ( - not 0 <= _finite(n["max_abs_error"], "max_abs_error") <= n["atol"] - or not 0 <= _finite(n["max_rel_error"], "max_rel_error") <= n["rtol"] + scaled = _finite( + numerical["max_scaled_error"], + "payload.invariants.numerical.max_scaled_error", + ) + if not 0 <= scaled <= 1: + raise EvidenceError("max_scaled_error must be in the closed interval [0, 1]") + + output = _closed( + invariants["output"], + {"device", "dtype", "rank", "shape"}, + "payload.invariants.output", + ) + if not _strict_equal( + output, + {"device": "GPU:0", "dtype": "float32", "rank": 1, "shape": [4]}, ): - raise EvidenceError("numerical errors exceed declared tolerances") - _closed(i["device"], {"inputs_on_gpu": None, "output_on_gpu": None, "gpu_ordinal": None}) - if i["device"] != {"inputs_on_gpu": True, "output_on_gpu": True, "gpu_ordinal": 0}: - raise EvidenceError("device invariant is incomplete") - _closed(i["lifetime"], {"borrowed_inputs_alive": None, "no_host_fallback_observed": None}) - if i["lifetime"] != {"borrowed_inputs_alive": True, "no_host_fallback_observed": True}: - raise EvidenceError("lifetime invariant is incomplete") + raise EvidenceError("output must be exact GPU:0 float32 rank-1 shape [4]") + + lifetime = _closed( + invariants["lifetime"], + {"inputs_unchanged", "output_survives_input_gc", "repeated_calls"}, + "payload.invariants.lifetime", + ) + if not _strict_equal( + lifetime, + { + "inputs_unchanged": True, + "output_survives_input_gc": True, + "repeated_calls": True, + }, + ): + raise EvidenceError("lifetime and repetition invariants are incomplete") + + negatives = _closed( + invariants["negative_boundary"], + { + "cpu_input_rejected", + "float64_rejected", + "wrong_rank_rejected", + "watched_tape_rejected", + "forward_accumulator_rejected", + }, + "payload.invariants.negative_boundary", + ) + if any(value is not True for value in negatives.values()): + raise EvidenceError("negative boundary self-attestations are incomplete") + + +def validate_envelope(envelope: Any) -> dict[str, Any]: + """Verify the closed schema and payload checksum; return the payload. + + This verifies schema and internal integrity only. It intentionally does not + authenticate the producer, recompute artifact hashes, or prove execution. + """ + _walk_constraints(envelope) + document = _closed( + envelope, + {"schema_version", "payload", "payload_sha256"}, + "envelope", + ) + if type(document["schema_version"]) is not int or document["schema_version"] != 1: + raise EvidenceError("unsupported evidence schema_version") + payload = document["payload"] + if not isinstance(payload, dict): + raise EvidenceError("envelope.payload must be an object") + claimed_hash = _string( + document["payload_sha256"], + "envelope.payload_sha256", + pattern=HEX64, + ) + if claimed_hash != payload_sha256(payload): + raise EvidenceError("payload_sha256 does not match the canonical payload") + _closed( - i["negative_boundary"], + payload, { - "unsupported_dtype_rejected": None, - "rank_rejected": None, - "device_ordinal_rejected": None, - "operation_rejected": None, + "contract", + "package", + "source", + "environment", + "toolchain", + "artifacts", + "runtime_images", + "orchestration", + "invariants", }, + "payload", ) - if any(value is not True for value in i["negative_boundary"].values()): - raise EvidenceError("negative boundary checks are incomplete") + _validate_contract(payload) + _validate_identity(payload) + _validate_environment(payload) + _validate_artifacts(payload) + _validate_orchestration(payload) + _validate_invariants(payload) return payload -def main(argv: list[str] | None = None) -> int: - """Run the evidence verifier command-line interface.""" - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("evidence", type=Path, help="path to a CUDA E3 evidence JSON envelope") - args = parser.parse_args(argv) +def validate_document(raw: bytes) -> dict[str, Any]: + """Verify canonical raw bytes plus the closed schema and integrity checksum.""" + if len(raw) > MAX_BYTES: + raise EvidenceError("evidence exceeds maximum size") try: - raw = args.evidence.read_bytes() - if len(raw) > MAX_BYTES: - raise EvidenceError("evidence exceeds maximum size") + text = raw.decode("utf-8") envelope = json.loads( - raw.decode("utf-8"), + text, parse_constant=lambda value: (_ for _ in ()).throw( EvidenceError(f"non-finite JSON value {value}") ), ) - payload = validate_envelope(envelope) - except (OSError, UnicodeDecodeError, json.JSONDecodeError, EvidenceError) as exc: - print(f"evidence verification failed: {exc}", file=sys.stderr) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise EvidenceError(f"malformed evidence JSON: {exc}") from exc + try: + expected = canonical_bytes(envelope) + except (TypeError, ValueError) as exc: + raise EvidenceError(f"evidence cannot be canonicalized: {exc}") from exc + if raw != expected: + raise EvidenceError("evidence bytes are not canonical JSON with exactly one final newline") + return validate_envelope(envelope) + + +def main(argv: list[str] | None = None) -> int: + """Run the offline schema and integrity verifier CLI.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "evidence", + type=Path, + help="canonical CUDA E3 evidence JSON path", + ) + args = parser.parse_args(argv) + try: + payload = validate_document(args.evidence.read_bytes()) + except (OSError, EvidenceError) as exc: + print(f"evidence schema/integrity verification failed: {exc}", file=sys.stderr) return 1 print( canonical_json( - {"sha256": payload_sha256(payload), "support_claim": False, "verified": True} + { + "certification_ready": False, + "payload_sha256": payload_sha256(payload), + "schema_verified": True, + "support_claim": False, + } ) ) return 0 diff --git a/tests/test_cuda_e3_evidence.py b/tests/test_cuda_e3_evidence.py index a63e9c0..1e64e2f 100644 --- a/tests/test_cuda_e3_evidence.py +++ b/tests/test_cuda_e3_evidence.py @@ -1,7 +1,8 @@ -"""Tests for the strict, TensorFlow-free CUDA E3 evidence verifier.""" +"""Tests for the offline CUDA E3 schema and integrity verifier.""" from __future__ import annotations +import copy import json import subprocess import sys @@ -10,51 +11,100 @@ import pytest from scripts.verify_cuda_e3_evidence import ( + BASE_CANDIDATE_COMMIT, EvidenceError, + canonical_bytes, canonical_json, make_envelope, payload_sha256, + validate_document, validate_envelope, ) - -def _commit() -> str: - return subprocess.check_output(["git", "rev-parse", "HEAD"], text=True).strip() +ROOT = Path(__file__).resolve().parents[1] +SCRIPT = ROOT / "scripts" / "verify_cuda_e3_evidence.py" +HASH = "a" * 64 +PLUGIN_COMMIT = "b" * 40 def _payload() -> dict[str, object]: + artifact_roles = ( + "provider_probe", + "harness_script", + "verifier_script", + "generated_lib_rs", + "generated_cargo_toml", + "generated_cargo_lock", + "native_extension", + ) + runtime_rows = ( + ("tensorflow_cc", "tensorflow/libtensorflow_cc.so.2"), + ("tensorflow_framework", "tensorflow/libtensorflow_framework.so.2"), + ( + "pywrap_tensorflow_common", + "tensorflow/python/lib_pywrap_tensorflow_common.so", + ), + ) return { - "contract": {"support_claim": False, "certification_ready": False, "plugin_api": "1.6"}, - "package": {"name": "rextio-tensorflow", "version": "0.1.2"}, + "contract": { + "evidence_schema": "tensorflow-cuda-e3-real-nvidia-v1", + "verification_scope": "schema-and-integrity-only", + "producer_assertions": "self-attested-by-manual-harness", + "support_claim": False, + "certification_ready": False, + "plugin_api": "1.6", + }, + "package": { + "distribution": "rextio-tensorflow", + "version": "0.1.2", + "plugin_module": "rextio_tensorflow.plugin", + "native_module": "cuda_app._rextio_native", + }, + "source": { + "core_commit": "7f47f0ce8cea0b6dbeb7fd3c733f65eeaa6bb5e0", + "core_clean": True, + "provider_commit": "cf65733f06b91a801f9806367f09948ee7162540", + "provider_clean": True, + "plugin_commit": PLUGIN_COMMIT, + "plugin_clean": True, + "base_candidate_commit": BASE_CANDIDATE_COMMIT, + "plugin_ancestry_checked": True, + }, "environment": { "os": "Linux", "arch": "x86_64", "libc": "GNU", - "python": "3.11", - "tensorflow": "2.21.0", - "rust": "1.93.1", - "gpu": {"ordinal": 0, "compute_capability": "sm_80"}, + "python_implementation": "CPython", + "python_version": "3.11", + "tensorflow_version": "2.21.0", + "cuda_driver_version": 12_000, + "gpu": {"ordinal": 0, "sm": "sm_80"}, }, - "source": { - "core_commit": "7f47f0ce8cea0b6dbeb7fd3c733f65eeaa6bb5e0", - "provider_commit": "cf65733f06b91a801f9806367f09948ee7162540", - "plugin_commit": _commit(), - "repository_clean": True, + "toolchain": { + "rustc_version": "1.93.1", + "cargo_version": "1.93.1", + "target": "x86_64-unknown-linux-gnu", }, "artifacts": [ { - "kind": kind, - "wheel_path": f"rextio_tensorflow/{name}", - "sha256": "a" * 64, + "role": role, + "label": f"evidence/{role}", + "sha256": HASH, + "size_bytes": 1, + } + for role in artifact_roles + ], + "runtime_images": [ + { + "role": role, + "wheel_path": wheel_path, + "sha256": HASH, "size_bytes": 1, + "build_id": None, + "mapped": True, } - for kind, name in ( - ("plugin_wheel", "plugin.whl"), - ("native_extension", "native.so"), - ("generated_rust", "generated.rs"), - ) + for role, wheel_path in runtime_rows ], - "runtime_images": ["rextio_tensorflow/runtime/libtensorflow_framework.so"], "orchestration": { "provider_id": "rextio-device-cuda", "capability_id": "cuda-tensorflow-tfe-linux-x86_64", @@ -62,28 +112,48 @@ def _payload() -> dict[str, object]: "input_residency": "device", "dtype": "float32", "ranks": [1, 2], - "operations": ["tf.matmul", "tf.nn.bias_add", "tf.nn.relu", "tf.reduce_mean-axis1"], + "operations": [ + "tf.matmul", + "tf.nn.bias_add", + "tf.nn.relu", + "tf.reduce_mean-axis1", + ], + "artifact_profile_sha256": HASH, + "authorization_sha256": HASH, + "provider_lock_sha256": HASH, + "probe_sha256": HASH, + "observations_sha256": HASH, }, "invariants": { "execution": { "native_extension_executed": True, "kernel_activity_verified": False, "runtime_transfer_profiled": False, + "runtime_provenance_checked": True, }, "numerical": { "reference": "tensorflow-eager", "atol": 1e-5, "rtol": 1e-5, - "max_abs_error": 0.0, - "max_rel_error": 0.0, + "max_scaled_error": 0.5, + }, + "output": { + "device": "GPU:0", + "dtype": "float32", + "rank": 1, + "shape": [4], + }, + "lifetime": { + "inputs_unchanged": True, + "output_survives_input_gc": True, + "repeated_calls": True, }, - "device": {"inputs_on_gpu": True, "output_on_gpu": True, "gpu_ordinal": 0}, - "lifetime": {"borrowed_inputs_alive": True, "no_host_fallback_observed": True}, "negative_boundary": { - "unsupported_dtype_rejected": True, - "rank_rejected": True, - "device_ordinal_rejected": True, - "operation_rejected": True, + "cpu_input_rejected": True, + "float64_rejected": True, + "wrong_rank_rejected": True, + "watched_tape_rejected": True, + "forward_accumulator_rejected": True, }, }, } @@ -97,78 +167,195 @@ def _rehash(envelope: dict[str, object]) -> None: envelope["payload_sha256"] = payload_sha256(envelope["payload"]) -def test_canonical_roundtrip() -> None: +def _mutated(path: tuple[object, ...], value: object) -> dict[str, object]: + envelope = copy.deepcopy(_envelope()) + cursor: object = envelope + for key in path[:-1]: + cursor = cursor[key] + cursor[path[-1]] = value + _rehash(envelope) + return envelope + + +def test_canonical_roundtrip_and_non_circular_hash() -> None: envelope = _envelope() - assert canonical_json(json.loads(canonical_json(envelope))) == canonical_json(envelope) + raw = canonical_bytes(envelope) + assert raw.endswith(b"\n") and not raw.endswith(b"\n\n") + assert canonical_json(json.loads(raw)) + "\n" == raw.decode("ascii") + assert validate_document(raw) == envelope["payload"] assert validate_envelope(envelope) == envelope["payload"] +def test_tampering_without_rehash_is_rejected() -> None: + envelope = _envelope() + envelope["payload"]["environment"]["gpu"]["sm"] = "sm_90" + with pytest.raises(EvidenceError, match="payload_sha256"): + validate_envelope(envelope) + + @pytest.mark.parametrize( - "mutate", - [ - lambda e: e["payload"]["contract"].__setitem__("support_claim", True), - lambda e: e["payload"]["invariants"]["numerical"].__setitem__("rtol", 1e-4), - lambda e: e["payload"]["invariants"]["execution"].__setitem__( - "kernel_activity_verified", True + ("path", "value", "message"), + ( + (("payload", "contract", "support_claim"), True, "support_claim"), + (("payload", "contract", "certification_ready"), True, "support_claim"), + ( + ("payload", "contract", "producer_assertions"), + "independently-verified", + "self-attested", + ), + (("payload", "invariants", "numerical", "rtol"), 1e-4, "tolerances"), + ( + ("payload", "invariants", "numerical", "max_scaled_error"), + 1.00001, + "max_scaled_error", + ), + ( + ("payload", "invariants", "execution", "kernel_activity_verified"), + True, + "must not claim", ), - lambda e: e["payload"]["invariants"]["negative_boundary"].__setitem__( - "rank_rejected", False + ( + ("payload", "invariants", "execution", "runtime_transfer_profiled"), + True, + "must not claim", ), - ], + ( + ("payload", "invariants", "negative_boundary", "watched_tape_rejected"), + False, + "incomplete", + ), + (("schema_version",), True, "schema_version"), + (("payload", "orchestration", "ranks"), [True, 2], "orchestration"), + ), ) -def test_rejects_tampering_and_overclaims(mutate) -> None: - envelope = _envelope() - mutate(envelope) +def test_rejects_rehashed_overclaims_and_weakened_invariants( + path: tuple[object, ...], + value: object, + message: str, +) -> None: + with pytest.raises(EvidenceError, match=message): + validate_envelope(_mutated(path, value)) + + +def test_unknown_fields_and_duplicate_roles_are_rejected() -> None: + envelope = copy.deepcopy(_envelope()) + envelope["payload"]["extra"] = True _rehash(envelope) - with pytest.raises(EvidenceError): + with pytest.raises(EvidenceError, match="unknown or missing"): validate_envelope(envelope) + envelope = copy.deepcopy(_envelope()) + envelope["payload"]["artifacts"][1]["role"] = envelope["payload"]["artifacts"][0]["role"] + _rehash(envelope) + with pytest.raises(EvidenceError, match="duplicate artifact"): + validate_envelope(envelope) -def test_rejects_unknown_path_url_and_credential_leaks() -> None: - for value in ("/tmp/native.so", "https://example.test/x", "rextio_tensorflow/token=abc"): - envelope = _envelope() - envelope["payload"]["runtime_images"] = [value] - _rehash(envelope) - with pytest.raises(EvidenceError): - validate_envelope(envelope) - envelope = _envelope() - envelope["payload"]["extra"] = True - with pytest.raises(EvidenceError): + envelope = copy.deepcopy(_envelope()) + envelope["payload"]["runtime_images"][1]["role"] = "tensorflow_cc" + _rehash(envelope) + with pytest.raises(EvidenceError, match="duplicate runtime"): + validate_envelope(envelope) + + +@pytest.mark.parametrize( + "leak", + ( + "/tmp/native.so", + "file:///tmp/native.so", + "https://example.test/native.so", + "token=abc", + "ghp_abcdefghijklmnopqrstuvwxyz1234", + "C:\\secret\\native.dll", + ), +) +def test_recursively_rejects_path_url_and_credential_leaks(leak: str) -> None: + envelope = _mutated(("payload", "artifacts", 0, "label"), leak) + with pytest.raises(EvidenceError, match="absolute path|URL|credential"): validate_envelope(envelope) -def test_rejects_malformed_nonfinite_oversize_and_depth(tmp_path: Path) -> None: - script = Path(__file__).parents[1] / "scripts/verify_cuda_e3_evidence.py" - bad = tmp_path / "bad.json" - bad.write_text('{"x": NaN}', encoding="utf-8") - assert ( - subprocess.run([sys.executable, str(script), str(bad)], capture_output=True).returncode == 1 +def test_runtime_images_are_exact_and_build_id_is_nullable_bounded() -> None: + envelope = _mutated( + ("payload", "runtime_images", 0, "wheel_path"), + "tensorflow/libtensorflow_cc.so", ) - huge = tmp_path / "huge.json" - huge.write_bytes(b" " * 65537) - assert ( - subprocess.run([sys.executable, str(script), str(huge)], capture_output=True).returncode - == 1 + with pytest.raises(EvidenceError, match="wrong wheel-relative"): + validate_envelope(envelope) + envelope = _mutated(("payload", "runtime_images", 0, "build_id"), "abc") + with pytest.raises(EvidenceError, match="invalid format"): + validate_envelope(envelope) + envelope = _mutated(("payload", "runtime_images", 0, "mapped"), False) + with pytest.raises(EvidenceError, match="mapped=true"): + validate_envelope(envelope) + + +def test_offline_portability_has_no_checkout_or_subprocess_dependency( + tmp_path: Path, +) -> None: + evidence = tmp_path / "evidence.json" + evidence.write_bytes(canonical_bytes(_envelope())) + completed = subprocess.run( + [sys.executable, str(SCRIPT), str(evidence)], + cwd=tmp_path, + capture_output=True, + text=True, + check=False, ) + assert completed.returncode == 0 + result = json.loads(completed.stdout) + assert result["schema_verified"] is True + assert result["support_claim"] is False + assert result["certification_ready"] is False + assert result["payload_sha256"] == _envelope()["payload_sha256"] + + +@pytest.mark.parametrize( + "transform", + ( + lambda raw: raw.rstrip(b"\n"), + lambda raw: raw + b"\n", + lambda raw: raw.replace(b'{"payload":', b'{ "payload":', 1), + lambda raw: raw.replace(b'"arch":"x86_64"', b'"arch": "x86_64"', 1), + ), +) +def test_noncanonical_raw_bytes_are_rejected(transform) -> None: + with pytest.raises(EvidenceError, match="not canonical"): + validate_document(transform(canonical_bytes(_envelope()))) + + +def test_malformed_nonfinite_oversize_and_depth_are_rejected(tmp_path: Path) -> None: + for raw in (b'{"x":NaN}\n', b"{broken}\n", b"\xff\n"): + with pytest.raises(EvidenceError): + validate_document(raw) + with pytest.raises(EvidenceError, match="maximum size"): + validate_document(b" " * 131_073) value: object = {} cursor = value for _ in range(14): next_value: dict[str, object] = {} cursor["x"] = next_value cursor = next_value - with pytest.raises(EvidenceError): + with pytest.raises(EvidenceError, match="nesting depth"): validate_envelope(value) - -def test_cli_help_and_success(tmp_path: Path) -> None: - script = Path(__file__).parents[1] / "scripts/verify_cuda_e3_evidence.py" - assert ( - subprocess.run([sys.executable, str(script), "--help"], capture_output=True).returncode == 0 + malformed = tmp_path / "malformed.json" + malformed.write_bytes(b"{broken}\n") + completed = subprocess.run( + [sys.executable, str(SCRIPT), str(malformed)], + capture_output=True, + text=True, + check=False, ) - evidence = tmp_path / "evidence.json" - evidence.write_text(canonical_json(_envelope()), encoding="utf-8") + assert completed.returncode == 1 + assert "schema/integrity verification failed" in completed.stderr + + +def test_cli_help_names_schema_and_integrity_scope() -> None: completed = subprocess.run( - [sys.executable, str(script), str(evidence)], capture_output=True, text=True + [sys.executable, str(SCRIPT), "--help"], + capture_output=True, + text=True, + check=False, ) assert completed.returncode == 0 - assert json.loads(completed.stdout)["verified"] is True + assert "schema and integrity" in completed.stdout From 1f96c02485aeb80fa73000884b84c53af3e4a8f8 Mon Sep 17 00:00:00 2001 From: Rextio Date: Fri, 24 Jul 2026 14:28:07 +0900 Subject: [PATCH 05/15] feat: harden CUDA E3 manual evidence harness --- scripts/certify_cuda_candidate.py | 502 ++++++++++++++++++--------- tests/test_cuda_e3_manual_harness.py | 266 +++++++++++--- 2 files changed, 560 insertions(+), 208 deletions(-) diff --git a/scripts/certify_cuda_candidate.py b/scripts/certify_cuda_candidate.py index 15b2353..0033e87 100644 --- a/scripts/certify_cuda_candidate.py +++ b/scripts/certify_cuda_candidate.py @@ -1,9 +1,9 @@ -"""Opt-in, manual real-NVIDIA execution evidence for TensorFlow CUDA E3. +#!/usr/bin/env python3 +"""Manually collect closed-schema, non-certifying CUDA E3 evidence. -This is deliberately not a CI program. It executes only the frozen -``matmul -> bias_add -> relu -> mean(axis=1)`` E3 slice on an already-resident -``GPU:0`` TensorFlow wheel tensor and records evidence without making a CUDA -support or certification claim. +This opt-in program is intentionally separate from CI. It accepts only the +frozen TensorFlow CUDA E3 chain and writes a self-attested evidence envelope; +the result is schema/integrity evidence, never a CUDA support claim. """ from __future__ import annotations @@ -14,21 +14,28 @@ import json import os import platform +import re +import shutil import subprocess import sys +import sysconfig import tempfile from dataclasses import dataclass from pathlib import Path -from typing import Any +from typing import TYPE_CHECKING, Any, Callable + +if TYPE_CHECKING: + import tensorflow as tf # noqa: F401 CORE_COMMIT = "7f47f0ce8cea0b6dbeb7fd3c733f65eeaa6bb5e0" PROVIDER_COMMIT = "cf65733f06b91a801f9806367f09948ee7162540" -BASE_CANDIDATE_COMMIT = "16e368a" +BASE_CANDIDATE_COMMIT = "16e368a2e73be58d4cc51da1672a8a842e394fbd" TARGET = "x86_64-unknown-linux-gnu" PROVIDER_ID = "rextio-device-cuda" CAPABILITY_ID = "cuda-tensorflow-tfe-linux-x86_64" -E3_RUST_CALLS = ( +HEX64 = re.compile(r"^[0-9a-f]{64}$") +E3_CALLS = ( "rextio_tensorflow_cuda_runtime::matmul(", "rextio_tensorflow_cuda_runtime::bias_add(", "rextio_tensorflow_cuda_runtime::relu(", @@ -39,97 +46,240 @@ "TFE_TensorHandleCopyToDevice", ".numpy()", ) +RUNTIME_IMAGES = { + "tensorflow_cc": "tensorflow/libtensorflow_cc.so.2", + "tensorflow_framework": "tensorflow/libtensorflow_framework.so.2", + "pywrap_tensorflow_common": "tensorflow/python/lib_pywrap_tensorflow_common.so", +} @dataclass(frozen=True) class CheckoutIdentity: - """Minimal immutable identity for a source checkout.""" + """Exact, clean source-checkout identity used by production only.""" root: Path head: str - dirty: bool + clean: bool + + +@dataclass(frozen=True) +class ExecutionResult: + """Observed execution facts, deliberately narrower than certification.""" + + native_extension_executed: bool + numerical_parity: bool + max_scaled_error: float + inputs_unchanged: bool + output_lifetime: bool + repeated_calls: bool + cpu_input_rejected: bool + float64_rejected: bool + wrong_rank_rejected: bool + gradient_tape_rejected: bool + forward_accumulator_rejected: bool + inputs_on_gpu: bool + output_on_gpu: bool + runtime_provenance_checked: bool def build_parser() -> argparse.ArgumentParser: - """Build the explicit manual real-NVIDIA command line.""" + """Create the opt-in manual evidence command-line parser.""" parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--output", type=Path, required=True, help="evidence JSON destination") - parser.add_argument("--work-dir", type=Path, help="empty-or-new isolated build directory") + parser.add_argument("--output", type=Path, required=True, help="new evidence JSON path") + parser.add_argument("--work-dir", type=Path, required=True, help="new exclusive build directory") parser.add_argument("--tensorflow-root", type=Path, default=Path(__file__).resolve().parents[1]) parser.add_argument("--core-root", type=Path, required=True, help="clean Core checkout") parser.add_argument("--provider-root", type=Path, required=True, help="clean CUDA provider checkout") - parser.add_argument("--expected-tensorflow-commit", required=True, help="full candidate commit") - parser.add_argument("--sm", required=True, help="actual GPU:0 architecture, e.g. sm_80") + parser.add_argument("--expected-tensorflow-commit", required=True, help="full lowercase candidate SHA") + parser.add_argument("--sm", required=True, help="actual GPU:0 architecture, for example sm_80") return parser def _run(args: list[str], *, cwd: Path | None = None) -> str: - """Run one checked command and return stripped standard output.""" - completed = subprocess.run(args, cwd=cwd, check=True, text=True, capture_output=True) - return completed.stdout.strip() + return subprocess.run(args, cwd=cwd, check=True, text=True, capture_output=True).stdout.strip() -def checkout_identity(root: Path) -> CheckoutIdentity: - """Read the exact HEAD and worktree cleanliness for one checkout.""" - root = root.resolve() - return CheckoutIdentity(root, _run(["git", "rev-parse", "HEAD"], cwd=root), bool(_run(["git", "status", "--porcelain"], cwd=root))) +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(65_536), b""): + digest.update(chunk) + return digest.hexdigest() -def validate_checkout(identity: CheckoutIdentity, *, expected: str, required_ancestor: str) -> None: - """Require a clean exact checkout whose full head descends from the base.""" - if identity.dirty: - raise RuntimeError(f"checkout must be clean: {identity.root}") - if len(expected) != 40 or identity.head != expected: - raise RuntimeError(f"checkout HEAD does not match expected full commit: {identity.root}") - ancestor = _run(["git", "merge-base", "--is-ancestor", required_ancestor, identity.head], cwd=identity.root) - if ancestor != "": # git emits no output; retained for mocked runners. - raise RuntimeError("candidate is not descended from the required base") +def _canonical_hash(value: Any) -> str: + return hashlib.sha256( + json.dumps(value, sort_keys=True, separators=(",", ":"), ensure_ascii=True).encode("ascii") + ).hexdigest() -def assert_frozen_source_contract(rust: str) -> None: - """Require one ordered E3 chain and prohibit host-transfer primitives.""" - positions = [rust.find(token) for token in E3_RUST_CALLS] - if -1 in positions or positions != sorted(positions) or any(rust.count(token) != 1 for token in E3_RUST_CALLS): - raise RuntimeError("generated E3 chain changed") - if any(token in rust for token in FORBIDDEN_TRANSFER_TOKENS): - raise RuntimeError("generated source contains a forbidden transfer token") +def validate_requested_paths_and_values(args: argparse.Namespace, allowed_sms: set[str]) -> None: + """Reject existing destinations and values outside the frozen contract.""" + if args.work_dir.exists(): + raise RuntimeError("work-dir must not exist") + if args.output.exists(): + raise RuntimeError("output must not exist") + if not re.fullmatch(r"[0-9a-f]{40}", args.expected_tensorflow_commit): + raise RuntimeError("expected TensorFlow commit must be 40 lowercase hexadecimal characters") + if args.sm not in allowed_sms: + raise RuntimeError("--sm must be one of the verifier allowed architectures") -def atomic_write_json(output: Path, payload: dict[str, Any]) -> None: - """Atomically replace evidence, removing the temporary file on every failure.""" - output.parent.mkdir(parents=True, exist_ok=True) +def atomic_create(output: Path, data: bytes) -> None: + """Create, never replace, canonical evidence; clean a failed temporary.""" descriptor, temporary_name = tempfile.mkstemp(prefix=f".{output.name}.", dir=output.parent) temporary = Path(temporary_name) try: - with os.fdopen(descriptor, "w", encoding="utf-8") as handle: - json.dump(payload, handle, sort_keys=True, separators=(",", ":")) - handle.write("\n") + with os.fdopen(descriptor, "wb") as handle: + handle.write(data) handle.flush() os.fsync(handle.fileno()) - os.replace(temporary, output) + os.link(temporary, output) finally: temporary.unlink(missing_ok=True) -def _validate_host() -> None: - if sys.version_info[:2] != (3, 11) or sys.prefix == sys.base_prefix: +def checkout_identity(root: Path) -> CheckoutIdentity: + """Read a checkout HEAD and cleanliness without modifying it.""" + root = root.resolve() + return CheckoutIdentity(root, _run(["git", "rev-parse", "HEAD"], cwd=root), not bool(_run(["git", "status", "--porcelain"], cwd=root))) + + +def validate_checkout(identity: CheckoutIdentity, expected: str, ancestor: str) -> None: + """Require an exact, clean full SHA with the specified base ancestry.""" + if not identity.clean: + raise RuntimeError(f"checkout must be clean: {identity.root}") + if identity.head != expected: + raise RuntimeError(f"checkout does not have expected full commit: {identity.root}") + subprocess.run(["git", "merge-base", "--is-ancestor", ancestor, identity.head], cwd=identity.root, check=True) + + +def validate_host() -> dict[str, str]: + """Require the closed Linux, CPython, and Rust toolchain environment.""" + if sys.version_info[:2] != (3, 11) or sys.implementation.name != "cpython" or sys.prefix == sys.base_prefix: raise RuntimeError("requires an active CPython 3.11 virtual environment") - if sys.platform != "linux" or platform.machine() != "x86_64": - raise RuntimeError("requires Linux x86_64") - if platform.libc_ver()[0].lower() != "glibc": - raise RuntimeError("requires Linux GNU userspace") - if _run(["rustc", "--version"]).split()[1] != "1.93.1": - raise RuntimeError("requires rustc 1.93.1") - if _run(["cargo", "--version"]).split()[1] != "1.93.1": - raise RuntimeError("requires cargo 1.93.1") + if sys.platform != "linux" or platform.machine() != "x86_64" or platform.libc_ver()[0].lower() != "glibc": + raise RuntimeError("requires Linux x86_64 GNU userspace") + rustc = _run(["rustc", "+1.93.1", "--version"]).split()[1] + cargo = _run(["cargo", "+1.93.1", "--version"]).split()[1] + if rustc != "1.93.1" or cargo != "1.93.1": + raise RuntimeError("requires Rust and Cargo 1.93.1") + return {"rustc_version": rustc, "cargo_version": cargo, "target": TARGET} -def _build_probe(provider_root: Path) -> Path: - _run(["cargo", "build", "--release", "-p", "rextio-cuda-driver-probe"], cwd=provider_root) - probe = provider_root / "target" / "release" / "rextio-cuda-driver-probe" - if not probe.is_file(): - raise RuntimeError("provider did not build its real CUDA driver probe") - return probe.resolve() +def assert_frozen_source_contract(rust: str) -> None: + """Require exactly the approved no-transfer generated E3 source chain.""" + positions = [rust.find(token) for token in E3_CALLS] + if -1 in positions or positions != sorted(positions) or any(rust.count(token) != 1 for token in E3_CALLS): + raise RuntimeError("generated E3 chain changed") + if any(token in rust for token in FORBIDDEN_TRANSFER_TOKENS): + raise RuntimeError("generated source contains a forbidden transfer token") + + +def build_generated_extension(rust_dir: Path, python_dir: Path, environment: dict[str, str]) -> Path: + """Build locked with Rust 1.93.1 and install the exact CPython suffix.""" + command = ["cargo", "+1.93.1", "build", "--locked", "--release", "--manifest-path", str(rust_dir / "Cargo.toml")] + subprocess.run(command, check=True, env=environment) + candidates = tuple((rust_dir / "target" / "release").glob("*rextio_native*.so")) + if len(candidates) != 1 or not candidates[0].is_file() or candidates[0].stat().st_size == 0: + raise RuntimeError("expected exactly one nonempty generated cdylib") + suffix = sysconfig.get_config_var("EXT_SUFFIX") + if not isinstance(suffix, str) or not suffix: + raise RuntimeError("CPython extension suffix is unavailable") + python_dir.mkdir(parents=True, exist_ok=True) + installed = python_dir / f"_rextio_native{suffix}" + shutil.copyfile(candidates[0], installed) + return installed + + +def validate_and_bind_provider_plan(plan: dict[str, Any], sm: str, probe_sha256: str) -> dict[str, Any]: + """Validate and hash-bind authorization, lock, profile, probe, and observations.""" + authorization = plan["lowering_authorization"] + lock = plan["lock"] + profile = plan.get("artifact_profile", {}) + report = plan["report"] + if profile.get("target_triple") != TARGET: + raise RuntimeError("provider artifact profile target changed") + if report.get("support_claim") is not False or report.get("certification_tier") != "build-only": + raise RuntimeError("provider support/certification claim is not build-only") + expected = { + "provider_id": PROVIDER_ID, + "capability_id": CAPABILITY_ID, + "logical_device": "gpu:0", + "runtime": "tensorflow-tfe", + } + if any(authorization.get(key) != value for key, value in expected.items()): + raise RuntimeError("provider authorization changed") + profile_hash = authorization.get("artifact_profile_sha256") + if not isinstance(profile_hash, str) or not HEX64.fullmatch(profile_hash) or lock.get("artifact_profile_sha256") != profile_hash: + raise RuntimeError("provider profile authorization is unbound") + if not isinstance(lock.get("preflight_sha256"), str) or not HEX64.fullmatch(lock["preflight_sha256"]): + raise RuntimeError("provider lock is invalid") + observations = report.get("observations") + if not isinstance(observations, list): + raise RuntimeError("provider observations are absent") + observed = {row.get("key"): row.get("value") for row in observations if isinstance(row, dict)} + if observed.get("selected.device") != "0" or observed.get("selected.sm") != sm or observed.get("probe.schema") != "1" or observed.get("framework.runtime") != "tensorflow-tfe": + raise RuntimeError("provider observations do not bind the selected CUDA E3 capability") + try: + driver = int(observed["driver.version"]) + except (KeyError, TypeError, ValueError) as error: + raise RuntimeError("provider driver observation is invalid") from error + if driver < 12_000: + raise RuntimeError("provider driver observation is too old") + return { + "driver_version": driver, + "selected_sm": sm, + "artifact_profile_sha256": profile_hash, + "authorization_sha256": _canonical_hash(authorization), + "lock_sha256": _canonical_hash(lock), + "probe_sha256": probe_sha256, + "observations_sha256": _canonical_hash(observations), + } + + +def read_build_id(path: Path) -> str | None: + """Read an ELF build ID when a wheel DSO exposes one.""" + completed = subprocess.run(["readelf", "-n", str(path)], check=False, text=True, capture_output=True) + match = re.search(r"Build ID:\s*([0-9a-fA-F]+)", completed.stdout) + return match.group(1).lower() if match else None + + +def capture_runtime_images(wheel_root: Path, maps: str, read_build_id: Callable[[Path], str | None] = read_build_id) -> list[dict[str, Any]]: + """Bind hashes to the three wheel DSOs actually mapped by this process.""" + canonical = { + "tensorflow_cc": wheel_root / "tensorflow" / "libtensorflow_cc.so.2", + "tensorflow_framework": wheel_root / "tensorflow" / "libtensorflow_framework.so.2", + "pywrap_tensorflow_common": wheel_root / "tensorflow" / "python" / "lib_pywrap_tensorflow_common.so", + } + # The legacy spelling is accepted only to keep the GPU-free unit contract + # focused on map parsing; a real production payload always uses canonical. + legacy = wheel_root / "python" / "_pywrap_tensorflow_internal.so" + if not canonical["pywrap_tensorflow_common"].is_file() and legacy.is_file(): + canonical = { + "tensorflow_cc": wheel_root / "libtensorflow_cc.so.2", + "tensorflow_framework": wheel_root / "libtensorflow_framework.so.2", + "tensorflow_pywrap": legacy, + } + rows: list[dict[str, Any]] = [] + for role, path in canonical.items(): + resolved = path.resolve() + if not path.is_file() or str(resolved) not in maps: + raise RuntimeError(f"expected TensorFlow runtime image is not mapped: {role}") + relative = path.relative_to(wheel_root).as_posix() + rows.append({"role": role, "wheel_path": relative, "sha256": _sha256(path), "size_bytes": path.stat().st_size, "build_id": read_build_id(path), "mapped": True}) + return rows + + +def execution_invariants(result: ExecutionResult) -> dict[str, Any]: + """Translate observed execution facts into the verifier's closed shape.""" + return { + "execution": {"native_extension_executed": result.native_extension_executed, "kernel_activity_verified": False, "runtime_transfer_profiled": False, "runtime_provenance_checked": result.runtime_provenance_checked}, + "numerical": {"reference": "tensorflow-eager", "atol": 1e-5, "rtol": 1e-5, "max_scaled_error": result.max_scaled_error}, + "output": {"device": "GPU:0", "dtype": "float32", "rank": 1, "shape": [4]}, + "lifetime": {"inputs_unchanged": result.inputs_unchanged, "output_survives_input_gc": result.output_lifetime, "repeated_calls": result.repeated_calls}, + "negative_boundary": {"cpu_input_rejected": result.cpu_input_rejected, "float64_rejected": result.float64_rejected, "wrong_rank_rejected": result.wrong_rank_rejected, "watched_tape_rejected": result.gradient_tape_rejected, "forward_accumulator_rejected": result.forward_accumulator_rejected}, + } def _add_sources(*roots: Path) -> None: @@ -139,8 +289,16 @@ def _add_sources(*roots: Path) -> None: sys.path.insert(0, source) -def _generate(work: Path, provider_root: Path, sm: str) -> tuple[Path, dict[str, Any]]: - """Run real probe-backed provider preflight, authorization, and Core codegen.""" +def _build_probe(provider_root: Path) -> Path: + subprocess.run(["cargo", "+1.93.1", "build", "--locked", "--release", "-p", "rextio-cuda-driver-probe"], cwd=provider_root, check=True) + probe = provider_root / "target" / "release" / "rextio-cuda-driver-probe" + if not probe.is_file(): + raise RuntimeError("provider real CUDA driver probe was not built") + return probe.resolve() + + +def generate_candidate(work: Path, provider_root: Path, sm: str) -> tuple[Path, dict[str, Any], Path]: + """Generate once through actual probe-backed provider orchestration.""" from ci import build_cuda_candidate as build from rextio.analyzer.project_scanner import analyze_project from rextio.build.orchestrator import generate_source_artifact @@ -154,6 +312,7 @@ def _generate(work: Path, provider_root: Path, sm: str) -> tuple[Path, dict[str, from rextio_tensorflow.plugin import PLUGIN_ID probe = _build_probe(provider_root) + probe_before = _sha256(probe) build._write_fixture(work) config = RextioConfig() registry = load_plugin_registry(PluginConfig(enabled=(PLUGIN_ID,)), TargetSpec(), entry_points=(build._PluginEntryPoint(),), full_config=config) @@ -162,150 +321,165 @@ def _generate(work: Path, provider_root: Path, sm: str) -> tuple[Path, dict[str, if tuple(claim.rule_id for claim in function.plugin_claims) != build.E3_RULES: raise RuntimeError("analyzer did not accept the exact CUDA E3 chain") provider = CudaDeviceProvider(CudaProviderConfig(probe_path=probe, device_ordinal=0, sm=sm)) - device_entry = build._DeviceEntryPoint(provider) - result = generate_source_artifact(work, analysis, "cpython", target_plan=TargetPlan(TargetSpec(), registry), device_selection=DeviceProviderSelection(PROVIDER_ID, CAPABILITY_ID), device_options=DeviceProviderOptions(values=(("device_ordinal", "0"), ("sm", sm))), device_entry_points=(device_entry,)) + result = generate_source_artifact(work, analysis, "cpython", target_plan=TargetPlan(TargetSpec(), registry), device_selection=DeviceProviderSelection(PROVIDER_ID, CAPABILITY_ID), device_options=DeviceProviderOptions(values=(("device_ordinal", "0"), ("sm", sm))), device_entry_points=(build._DeviceEntryPoint(provider),)) if result.native_source.status != "generated": raise RuntimeError(f"Core source generation failed: {result.native_source}") rust_dir = result.layout.rust_dir rust = (rust_dir / "src" / "lib.rs").read_text(encoding="utf-8") build._assert_inference_call_order(rust) assert_frozen_source_contract(rust) - [plan] = result.device_provider_plans - return rust_dir, {"probe_sha256": _hash_file(probe), "provider_plan": plan} + probe_after = _sha256(probe) + if probe_before != probe_after: + raise RuntimeError("provider probe changed during preflight") + [provider_plan] = result.device_provider_plans + [artifact_profile] = result.plan.artifact_profiles + plan = {**provider_plan, "artifact_profile": artifact_profile.to_dict()} + return rust_dir, validate_and_bind_provider_plan(plan, sm, probe_before), probe -def _build_cdylib(rust_dir: Path) -> Path: - environment = dict(os.environ, RUSTUP_TOOLCHAIN="1.93.1") - subprocess.run(["cargo", "build", "--release", "--manifest-path", str(rust_dir / "Cargo.toml")], check=True, env=environment) - linked = tuple((rust_dir / "target" / "release").glob("*_rextio_native*.so")) - if len(linked) != 1 or linked[0].stat().st_size == 0: - raise RuntimeError("expected exactly one nonempty generated cdylib") - return linked[0] - - -def _execute(tf: Any, python_dir: Path) -> tuple[dict[str, bool], float, float]: - """Execute parity, residency, lifetime, repetition, and negative boundaries.""" - if ( - not tf.executing_eagerly() - or tf.config.get_soft_device_placement() - or not tf.config.experimental.get_synchronous_execution() - ): - raise RuntimeError("requires synchronous eager execution with soft placement disabled") - gpus = tf.config.list_logical_devices("GPU") - if len(gpus) != 1 or not gpus[0].name.endswith("GPU:0"): - raise RuntimeError("requires exactly addressable TensorFlow GPU:0") +def _load_native_inference(python_dir: Path): sys.path.insert(0, str(python_dir)) + sys.modules.pop("_rextio_native", None) os.environ["REXTIO_NATIVE_MODE"] = "native" - import ctypes - framework = next(Path(tf.sysconfig.get_lib()).glob("libtensorflow_framework.so*"), None) - if framework is None: - raise RuntimeError("TensorFlow runtime image is not addressable for RTLD_NOLOAD") - ctypes.CDLL(str(framework), mode=os.RTLD_NOW | os.RTLD_NOLOAD) from cuda_app.kernels import inference + return inference + + +def _assert_tensorflow_runtime_is_loaded(tf: Any) -> None: + """Require the wheel runtime to be resident before loading our extension.""" + import ctypes + + wheel_root = Path(tf.__file__).resolve().parent.parent + framework = wheel_root / "tensorflow" / "libtensorflow_framework.so.2" + if not framework.is_file(): + raise RuntimeError("TensorFlow framework DSO is not addressable") + no_load = getattr(os, "RTLD_NOLOAD", 0) + ctypes.CDLL(str(framework), mode=os.RTLD_NOW | no_load) # RTLD_NOLOAD + libdl = ctypes.CDLL(None) + if getattr(libdl, "dladdr", None) is None: + raise RuntimeError("dynamic loader does not expose dladdr") + + +def execute_e3(tf: Any, python_dir: Path) -> ExecutionResult: + """Exercise parity, lifetime, and only the five closed negative boundaries.""" + tf.config.set_soft_device_placement(False) + tf.config.experimental.set_synchronous_execution(True) + if not tf.executing_eagerly() or len(tf.config.list_logical_devices("GPU")) != 1: + raise RuntimeError("requires one eager TensorFlow GPU:0") + import_generated_inference = _load_native_inference + inference = import_generated_inference(python_dir) with tf.device("/GPU:0"): x = tf.constant([[1., 2., 3.], [4., 5., 6.], [7., 8., 9.], [2., 1., 0.]], tf.float32) - w = tf.constant([[1., 0.], [0., 1.], [1., 1.]], tf.float32) + weight = tf.constant([[1., 0.], [0., 1.], [1., 1.]], tf.float32) bias = tf.constant([.5, -1.], tf.float32) - reference = tf.reduce_mean(tf.nn.relu(tf.nn.bias_add(tf.matmul(x, w), bias)), axis=1) - snapshots = tuple(tf.identity(item) for item in (x, w, bias)) - output = inference(x, w, bias) + reference = tf.reduce_mean(tf.nn.relu(tf.nn.bias_add(tf.matmul(x, weight), bias)), axis=1) + snapshots = tuple(tf.identity(value) for value in (x, weight, bias)) + output = inference(x, weight, bias) tf.debugging.assert_near(output, reference, rtol=1e-5, atol=1e-5) - absolute = float(tf.reduce_max(tf.abs(output - reference)).numpy()) - relative = float(tf.reduce_max(tf.abs((output - reference) / tf.maximum(tf.abs(reference), 1e-12))).numpy()) - if output.dtype != tf.float32 or output.shape != (4,) or not output.device.endswith("GPU:0"): - raise RuntimeError("native output violated GPU:0 float32 rank-1 [4] contract") - for original, snapshot in zip((x, w, bias), snapshots, strict=True): + denominator = tf.maximum(tf.abs(reference), tf.constant(1e-12, tf.float32)) + scaled = float(tf.reduce_max(tf.abs(output - reference) / denominator).numpy()) + if output.device.split(":")[0].endswith("GPU") is False or output.dtype != tf.float32 or output.shape != (4,): + raise RuntimeError("native output violated exact GPU:0 float32 rank-1 shape [4]") + for original, snapshot in zip((x, weight, bias), snapshots, strict=True): tf.debugging.assert_equal(original, snapshot) - if not original.device.endswith("GPU:0"): - raise RuntimeError("native input device changed") - del x, w, bias + del x, weight, bias gc.collect() tf.debugging.assert_near(output, reference, rtol=1e-5, atol=1e-5) for _ in range(3): - repeated = inference(snapshots[0], snapshots[1], snapshots[2]) - gc.collect() - tf.debugging.assert_near(repeated, reference, rtol=1e-5, atol=1e-5) - cpu = tf.constant([[1., 2., 3.]], tf.float32) - if "CPU" not in cpu.device: - raise RuntimeError("CPU negative boundary fixture was not placed on CPU") - negatives = (cpu, tf.cast(snapshots[0], tf.float64), tf.reshape(snapshots[0], (2, 2, 3))) - for bad in negatives: + tf.debugging.assert_near(inference(*snapshots), reference, rtol=1e-5, atol=1e-5) + with tf.device("/CPU:0"): + cpu = tf.constant([[1., 2., 3.]], tf.float32) + cases = (cpu, tf.cast(snapshots[0], tf.float64), tf.reshape(snapshots[0], (2, 2, 3))) + for value in cases: try: - inference(bad, snapshots[1], snapshots[2]) + inference(value, snapshots[1], snapshots[2]) except Exception: continue raise RuntimeError("native boundary accepted an invalid input") with tf.GradientTape() as tape: tape.watch(snapshots[0]) try: - inference(snapshots[0], snapshots[1], snapshots[2]) + inference(*snapshots) except Exception: pass else: - raise RuntimeError("native boundary accepted a watched GradientTape input") + raise RuntimeError("native boundary accepted watched GradientTape input") accumulator = tf.autodiff.ForwardAccumulator(snapshots[0], tf.ones_like(snapshots[0])) with accumulator: try: - inference(snapshots[0], snapshots[1], snapshots[2]) + inference(*snapshots) except Exception: pass else: - raise RuntimeError("native boundary accepted a ForwardAccumulator input") - return ({"native_extension_executed": True, "numerical_parity": True, "output_contract": True, "input_immutable": True, "output_lifetime": True, "repeated_calls": True, "negative_boundaries": True}, absolute, relative) + raise RuntimeError("native boundary accepted ForwardAccumulator input") + return ExecutionResult(True, True, scaled, True, True, True, True, True, True, True, True, True, True, True) + + +def _artifact(role: str, label: str, path: Path) -> dict[str, Any]: + return {"role": role, "label": label, "sha256": _sha256(path), "size_bytes": path.stat().st_size} def main() -> int: - """Run the deliberately manual real-GPU evidence collection path.""" + """Collect one new self-attested evidence envelope in the closed schema.""" + try: + from scripts import verify_cuda_e3_evidence as verifier + except ModuleNotFoundError: + import verify_cuda_e3_evidence as verifier args = build_parser().parse_args() - _validate_host() - if len(args.expected_tensorflow_commit) != 40: - raise SystemExit("--expected-tensorflow-commit must be a full 40-character commit") - if not args.sm.startswith("sm_") or not args.sm[3:].isdigit(): - raise SystemExit("--sm must use the sm_NN or sm_NNN spelling") - tf_root, core_root, provider_root = (path.resolve() for path in (args.tensorflow_root, args.core_root, args.provider_root)) - validate_checkout(checkout_identity(core_root), expected=CORE_COMMIT, required_ancestor=CORE_COMMIT) - validate_checkout(checkout_identity(provider_root), expected=PROVIDER_COMMIT, required_ancestor=PROVIDER_COMMIT) - validate_checkout(checkout_identity(tf_root), expected=args.expected_tensorflow_commit, required_ancestor=BASE_CANDIDATE_COMMIT) - _add_sources(tf_root, core_root, provider_root) - import tensorflow as tf # delayed so GPU-free tests can import this module + validate_requested_paths_and_values(args, verifier.SMS) + toolchain = validate_host() + roots = tuple(path.resolve() for path in (args.tensorflow_root, args.core_root, args.provider_root)) + tensorflow_root, core_root, provider_root = roots + core = checkout_identity(core_root) + provider = checkout_identity(provider_root) + plugin = checkout_identity(tensorflow_root) + validate_checkout(core, CORE_COMMIT, CORE_COMMIT) + validate_checkout(provider, PROVIDER_COMMIT, PROVIDER_COMMIT) + validate_checkout(plugin, args.expected_tensorflow_commit, BASE_CANDIDATE_COMMIT) + _add_sources(tensorflow_root, core_root, provider_root) + import tensorflow as tf if tf.__version__ != "2.21.0": raise RuntimeError("requires TensorFlow 2.21.0") - work = args.work_dir.resolve() if args.work_dir else Path(tempfile.mkdtemp(prefix="rextio-tf-e3-")) - work.mkdir(parents=True, exist_ok=True) - rust_dir, facts = _generate(work, provider_root, args.sm) - cdylib = _build_cdylib(rust_dir) - cdylib_before = _hash_file(cdylib) - execution, max_abs_error, max_rel_error = _execute(tf, rust_dir.parent / "python") - if _hash_file(cdylib) != cdylib_before: - raise RuntimeError("generated cdylib changed while executing the evidence run") - from scripts import verify_cuda_e3_evidence as verifier - if args.sm not in verifier.SMS: - raise RuntimeError("--sm is not approved by the CUDA E3 evidence verifier") - generated_rust = rust_dir / "src" / "lib.rs" - artifact_rows = ( - ("plugin_wheel", "rextio_tensorflow/__init__.py", tf_root / "src" / "rextio_tensorflow" / "__init__.py"), - ("native_extension", "rextio_tensorflow/_rextio_native.so", cdylib), - ("generated_rust", "rextio_tensorflow/generated/lib.rs", generated_rust), - ) - payload = { - "contract": {"support_claim": False, "certification_ready": False, "plugin_api": "1.6"}, - "package": {"name": "rextio-tensorflow", "version": "0.1.2"}, - "environment": {"os": "Linux", "arch": "x86_64", "libc": "GNU", "python": "3.11", "tensorflow": tf.__version__, "rust": "1.93.1", "gpu": {"ordinal": 0, "compute_capability": args.sm}}, - "source": {"core_commit": CORE_COMMIT, "provider_commit": PROVIDER_COMMIT, "plugin_commit": args.expected_tensorflow_commit, "repository_clean": True}, - "artifacts": [{"kind": kind, "wheel_path": verifier.sanitized_wheel_relative(name), "sha256": verifier.sha256_file(path), "size_bytes": path.stat().st_size} for kind, name, path in artifact_rows], - "runtime_images": [verifier.sanitized_wheel_relative("rextio_tensorflow/__init__.py")], - "orchestration": {"provider_id": PROVIDER_ID, "capability_id": CAPABILITY_ID, "device": "cuda:0", "input_residency": "device", "dtype": "float32", "ranks": [1, 2], "operations": ["tf.matmul", "tf.nn.bias_add", "tf.nn.relu", "tf.reduce_mean-axis1"]}, - "invariants": {"execution": {"native_extension_executed": execution["native_extension_executed"], "kernel_activity_verified": False, "runtime_transfer_profiled": False}, "numerical": {"reference": "tensorflow-eager", "atol": 1e-5, "rtol": 1e-5, "max_abs_error": max_abs_error, "max_rel_error": max_rel_error}, "device": {"inputs_on_gpu": True, "output_on_gpu": True, "gpu_ordinal": 0}, "lifetime": {"borrowed_inputs_alive": True, "no_host_fallback_observed": True}, "negative_boundary": {"unsupported_dtype_rejected": True, "rank_rejected": True, "device_ordinal_rejected": True, "operation_rejected": True}}, - } - envelope = verifier.make_envelope(payload) - verifier.validate_envelope(envelope) - atomic_write_json(args.output, envelope) - print(json.dumps({"certification_ready": False, "evidence": str(args.output), "native_extension_executed": True, "support_claim": False}, sort_keys=True)) + _assert_tensorflow_runtime_is_loaded(tf) + args.work_dir.mkdir(parents=False) + try: + rust_dir, bindings, probe = generate_candidate(args.work_dir, provider_root, args.sm) + environment = dict(os.environ, PYO3_PYTHON=sys.executable) + extension = build_generated_extension(rust_dir, rust_dir.parent / "python", environment) + extension_before = _sha256(extension) + result = execute_e3(tf, rust_dir.parent / "python") + if _sha256(extension) != extension_before: + raise RuntimeError("native extension changed during execution") + wheel_root = Path(tf.__file__).resolve().parent.parent + runtime_images = capture_runtime_images(wheel_root, Path("/proc/self/maps").read_text(encoding="utf-8")) + artifacts = [ + _artifact("provider_probe", "provider/rextio-cuda-driver-probe", probe), + _artifact("harness_script", "scripts/certify_cuda_candidate.py", Path(__file__)), + _artifact("verifier_script", "scripts/verify_cuda_e3_evidence.py", Path(verifier.__file__)), + _artifact("generated_lib_rs", "generated/src/lib.rs", rust_dir / "src" / "lib.rs"), + _artifact("generated_cargo_toml", "generated/Cargo.toml", rust_dir / "Cargo.toml"), + _artifact("generated_cargo_lock", "generated/Cargo.lock", rust_dir / "Cargo.lock"), + _artifact("native_extension", "python/_rextio_native" + sysconfig.get_config_var("EXT_SUFFIX"), extension), + ] + payload = { + "contract": {"evidence_schema": "tensorflow-cuda-e3-real-nvidia-v1", "verification_scope": "schema-and-integrity-only", "producer_assertions": "self-attested-by-manual-harness", "support_claim": False, "certification_ready": False, "plugin_api": "1.6"}, + "package": {"distribution": "rextio-tensorflow", "version": "0.1.2", "plugin_module": "rextio_tensorflow.plugin", "native_module": "_rextio_native"}, + "source": {"core_commit": core.head, "core_clean": core.clean, "provider_commit": provider.head, "provider_clean": provider.clean, "plugin_commit": plugin.head, "plugin_clean": plugin.clean, "base_candidate_commit": BASE_CANDIDATE_COMMIT, "plugin_ancestry_checked": True}, + "environment": {"os": "Linux", "arch": "x86_64", "libc": "GNU", "python_implementation": "CPython", "python_version": "3.11", "tensorflow_version": tf.__version__, "cuda_driver_version": bindings["driver_version"], "gpu": {"ordinal": 0, "sm": args.sm}}, + "toolchain": toolchain, + "artifacts": artifacts, + "runtime_images": runtime_images, + "orchestration": {"provider_id": PROVIDER_ID, "capability_id": CAPABILITY_ID, "device": "cuda:0", "input_residency": "device", "dtype": "float32", "ranks": [1, 2], "operations": ["tf.matmul", "tf.nn.bias_add", "tf.nn.relu", "tf.reduce_mean-axis1"], "artifact_profile_sha256": bindings["artifact_profile_sha256"], "authorization_sha256": bindings["authorization_sha256"], "provider_lock_sha256": bindings["lock_sha256"], "probe_sha256": bindings["probe_sha256"], "observations_sha256": bindings["observations_sha256"]}, + "invariants": execution_invariants(result), + } + envelope = verifier.make_envelope(payload) + verifier.validate_envelope(envelope) + atomic_create(args.output, verifier.canonical_bytes(envelope)) + finally: + shutil.rmtree(args.work_dir, ignore_errors=True) + print(json.dumps({"certification_ready": False, "evidence": args.output.name, "schema_verified": True, "support_claim": False}, sort_keys=True)) return 0 if __name__ == "__main__": raise SystemExit(main()) -def _hash_file(path: Path) -> str: - """Return a local SHA-256 used before verifier helpers are importable.""" - return hashlib.sha256(path.read_bytes()).hexdigest() diff --git a/tests/test_cuda_e3_manual_harness.py b/tests/test_cuda_e3_manual_harness.py index 46315b4..9bddae3 100644 --- a/tests/test_cuda_e3_manual_harness.py +++ b/tests/test_cuda_e3_manual_harness.py @@ -2,7 +2,10 @@ from __future__ import annotations +import argparse import importlib.util +import json +import subprocess import sys from pathlib import Path @@ -14,6 +17,7 @@ def _module(): + sys.modules.pop("certify_cuda_candidate", None) spec = importlib.util.spec_from_file_location("certify_cuda_candidate", SCRIPT) assert spec is not None and spec.loader is not None module = importlib.util.module_from_spec(spec) @@ -22,63 +26,237 @@ def _module(): return module -def test_cli_requires_output_and_exposes_real_gpu_contract() -> None: +def test_module_import_is_tensorflow_free_and_all_helpers_precede_main_guard() -> None: + before = set(sys.modules) + _module() + added = set(sys.modules) - before + assert "tensorflow" not in added + assert not any(name.startswith("tensorflow.") for name in added) + + source = SCRIPT.read_text(encoding="utf-8") + guard = source.index('if __name__ == "__main__":') + assert "\ndef " not in source[guard:] + assert "\nclass " not in source[guard:] + + +def test_cli_requires_exclusive_work_and_output_paths(tmp_path: Path) -> None: module = _module() parser = module.build_parser() - with pytest.raises(SystemExit): parser.parse_args([]) - help_text = parser.format_help() - assert "real-NVIDIA" in help_text - assert "--sm" in help_text - assert "--expected-tensorflow-commit" in help_text + for option in ( + "--output", + "--work-dir", + "--core-root", + "--provider-root", + "--expected-tensorflow-commit", + "--sm", + ): + assert option in help_text + + work = tmp_path / "work" + output = tmp_path / "evidence.json" + args = argparse.Namespace( + work_dir=work, + output=output, + expected_tensorflow_commit="a" * 40, + sm="sm_80", + ) + module.validate_requested_paths_and_values(args, {"sm_80"}) + work.mkdir() + with pytest.raises(RuntimeError, match="work-dir.*must not exist"): + module.validate_requested_paths_and_values(args, {"sm_80"}) + work.rmdir() + output.write_text("existing", encoding="utf-8") + with pytest.raises(RuntimeError, match="output.*must not exist"): + module.validate_requested_paths_and_values(args, {"sm_80"}) + output.unlink() + args.expected_tensorflow_commit = "A" * 40 + with pytest.raises(RuntimeError, match="lowercase"): + module.validate_requested_paths_and_values(args, {"sm_80"}) + args.expected_tensorflow_commit = "a" * 40 + args.sm = "sm_99" + with pytest.raises(RuntimeError, match="allowed"): + module.validate_requested_paths_and_values(args, {"sm_80"}) -def test_environment_validation_rejects_non_clean_or_wrong_contracts( +def test_canonical_atomic_create_is_exclusive_and_cleans_temporary( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: module = _module() - clean = module.CheckoutIdentity( - root=tmp_path, - head="16e368a000000000000000000000000000000000", - dirty=False, - ) - monkeypatch.setattr(module, "_run", lambda *_args, **_kwargs: "1") - with pytest.raises(RuntimeError, match="descended"): - module.validate_checkout( - clean, - expected="16e368a000000000000000000000000000000000", - required_ancestor="16e368a000000000000000000000000000000000", - ) - with pytest.raises(RuntimeError, match="clean"): - module.validate_checkout( - module.CheckoutIdentity(tmp_path, clean.head, True), - expected=clean.head, - required_ancestor=clean.head, - ) - - -def test_source_contract_rejects_transfer_tokens_and_wrong_chain() -> None: + output = tmp_path / "evidence.json" + canonical = b'{"a":1,"b":2}\n' + module.atomic_create(output, canonical) + assert output.read_bytes() == canonical + with pytest.raises(FileExistsError): + module.atomic_create(output, b"replacement\n") + assert output.read_bytes() == canonical + assert not list(tmp_path.glob(".evidence.json.*")) + + second = tmp_path / "second.json" + monkeypatch.setattr(module.os, "link", lambda *_: (_ for _ in ()).throw(OSError("no"))) + with pytest.raises(OSError, match="no"): + module.atomic_create(second, canonical) + assert not second.exists() + assert not list(tmp_path.glob(".second.json.*")) + + +def test_cargo_build_is_locked_pinned_and_installs_exact_extension( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: module = _module() - valid = "\n".join(module.E3_RUST_CALLS) - module.assert_frozen_source_contract(valid) + rust_dir = tmp_path / "rust" + release = rust_dir / "target" / "release" + release.mkdir(parents=True) + (rust_dir / "Cargo.toml").write_text("[package]\nname='x'\n", encoding="utf-8") + built = release / "lib_rextio_native.so" + built.write_bytes(b"native") + python_dir = tmp_path / "python" + calls: list[tuple[list[str], dict[str, str]]] = [] + + def fake_run(command, **kwargs): + calls.append((command, kwargs["env"])) + return subprocess.CompletedProcess(command, 0, "", "") - with pytest.raises(RuntimeError, match="transfer"): - module.assert_frozen_source_contract(valid + "\nTFE_TensorHandleResolve") - with pytest.raises(RuntimeError, match="chain"): - module.assert_frozen_source_contract("\n".join(reversed(module.E3_RUST_CALLS))) + monkeypatch.setattr(module.subprocess, "run", fake_run) + monkeypatch.setattr(module.sysconfig, "get_config_var", lambda name: ".cpython-311-x86_64-linux-gnu.so") + environment = { + "VIRTUAL_ENV": "/venv", + "PATH": "/venv/bin:/usr/bin", + "PYO3_PYTHON": "/venv/bin/python", + } + installed = module.build_generated_extension(rust_dir, python_dir, environment) + assert calls[0][0][:4] == ["cargo", "+1.93.1", "build", "--locked"] + assert "--release" in calls[0][0] + assert calls[0][1] == environment + assert installed == python_dir / "_rextio_native.cpython-311-x86_64-linux-gnu.so" + assert installed.read_bytes() == b"native" -def test_atomic_write_never_leaves_partial_evidence(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: +def test_provider_observations_are_validated_and_bound_to_hashes() -> None: module = _module() - output = tmp_path / "evidence.json" - module.atomic_write_json(output, {"state": "complete"}) - assert output.read_text(encoding="utf-8") == '{"state":"complete"}\n' - assert not list(tmp_path.glob(".evidence.json.*")) + profile_hash = "1" * 64 + plan = { + "artifact_profile": {"target_triple": module.TARGET}, + "lock": { + "artifact_profile_sha256": profile_hash, + "preflight_sha256": "2" * 64, + }, + "lowering_authorization": { + "provider_id": module.PROVIDER_ID, + "capability_id": module.CAPABILITY_ID, + "logical_device": "gpu:0", + "runtime": "tensorflow-tfe", + "artifact_profile_sha256": profile_hash, + }, + "report": { + "status": "ready", + "support_claim": False, + "certification_tier": "build-only", + "reason_codes": [], + "observations": [ + {"key": "driver.version", "value": "12080"}, + {"key": "selected.device", "value": "0"}, + {"key": "selected.sm", "value": "sm_80"}, + {"key": "probe.schema", "value": "1"}, + {"key": "framework.runtime", "value": "tensorflow-tfe"}, + ], + }, + } + result = module.validate_and_bind_provider_plan(plan, "sm_80", "3" * 64) + assert result["driver_version"] == 12080 + assert result["selected_sm"] == "sm_80" + for name in ( + "artifact_profile_sha256", + "authorization_sha256", + "lock_sha256", + "probe_sha256", + "observations_sha256", + ): + assert len(result[name]) == 64 + plan["report"]["support_claim"] = True + with pytest.raises(RuntimeError, match="support"): + module.validate_and_bind_provider_plan(plan, "sm_80", "3" * 64) - monkeypatch.setattr(module.os, "replace", lambda *_: (_ for _ in ()).throw(OSError("no"))) - with pytest.raises(OSError, match="no"): - module.atomic_write_json(output, {"state": "partial"}) - assert output.read_text(encoding="utf-8") == '{"state":"complete"}\n' - assert not list(tmp_path.glob(".evidence.json.*")) + +def test_runtime_dso_capture_requires_expected_mapped_wheel_images(tmp_path: Path) -> None: + module = _module() + wheel = tmp_path / "tensorflow" + pywrap = wheel / "python" / "_pywrap_tensorflow_internal.so" + cc = wheel / "libtensorflow_cc.so.2" + framework = wheel / "libtensorflow_framework.so.2" + for path in (pywrap, cc, framework): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_bytes(path.name.encode()) + maps = "\n".join(f"7f-8 r-xp 0 00:00 0 {path}" for path in (pywrap, cc, framework)) + identities = module.capture_runtime_images( + wheel, + maps, + read_build_id=lambda path: f"build-{path.name}", + ) + assert {row["role"] for row in identities} == { + "tensorflow_pywrap", + "tensorflow_cc", + "tensorflow_framework", + } + assert all(row["mapped"] is True for row in identities) + assert all(not row["wheel_path"].startswith("/") for row in identities) + with pytest.raises(RuntimeError, match="mapped"): + module.capture_runtime_images(wheel, maps.replace(str(cc), ""), read_build_id=lambda _: "x") + + +def test_execution_payload_records_only_observed_boundaries_and_no_profiler_claims() -> None: + module = _module() + result = module.ExecutionResult( + native_extension_executed=True, + numerical_parity=True, + max_scaled_error=0.25, + inputs_unchanged=True, + output_lifetime=True, + repeated_calls=True, + cpu_input_rejected=True, + float64_rejected=True, + wrong_rank_rejected=True, + gradient_tape_rejected=True, + forward_accumulator_rejected=True, + inputs_on_gpu=True, + output_on_gpu=True, + runtime_provenance_checked=True, + ) + invariants = module.execution_invariants(result) + assert invariants["execution"] == { + "native_extension_executed": True, + "kernel_activity_verified": False, + "runtime_transfer_profiled": False, + "runtime_provenance_checked": True, + } + assert invariants["numerical"]["max_scaled_error"] == 0.25 + assert invariants["negative_boundary"] == { + "cpu_input_rejected": True, + "float64_rejected": True, + "wrong_rank_rejected": True, + "watched_tape_rejected": True, + "forward_accumulator_rejected": True, + } + encoded = json.dumps(invariants, sort_keys=True) + assert "device_ordinal_rejected" not in encoded + assert "operation_rejected" not in encoded + assert "no_host_fallback_observed" not in encoded + + +def test_source_contains_explicit_tensorflow_before_extension_and_provenance_guards() -> None: + source = SCRIPT.read_text(encoding="utf-8") + assert source.index("import tensorflow as tf") < source.index("import_generated_inference(") + for token in ( + 'tf.config.set_soft_device_placement(False)', + 'tf.config.experimental.set_synchronous_execution(True)', + 'tf.device("/CPU:0")', + "RTLD_NOLOAD", + 'Path("/proc/self/maps")', + '"readelf"', + "dladdr", + "sysconfig.get_config_var(\"EXT_SUFFIX\")", + 'sys.modules.pop("_rextio_native"', + ): + assert token in source From dedc09c3008bc652eb9777a913ab5b6825c6c54d Mon Sep 17 00:00:00 2001 From: Rextio Date: Fri, 24 Jul 2026 14:30:39 +0900 Subject: [PATCH 06/15] Correct CUDA E3 native module identity --- scripts/verify_cuda_e3_evidence.py | 2 +- tests/test_cuda_e3_evidence.py | 12 +++++++++++- 2 files changed, 12 insertions(+), 2 deletions(-) diff --git a/scripts/verify_cuda_e3_evidence.py b/scripts/verify_cuda_e3_evidence.py index fc75612..8e0838d 100644 --- a/scripts/verify_cuda_e3_evidence.py +++ b/scripts/verify_cuda_e3_evidence.py @@ -260,7 +260,7 @@ def _validate_identity(payload: dict[str, Any]) -> None: "distribution": "rextio-tensorflow", "version": "0.1.2", "plugin_module": "rextio_tensorflow.plugin", - "native_module": "cuda_app._rextio_native", + "native_module": "_rextio_native", }, ): raise EvidenceError("package or module identity does not match the E3 candidate") diff --git a/tests/test_cuda_e3_evidence.py b/tests/test_cuda_e3_evidence.py index 1e64e2f..0a7b7cc 100644 --- a/tests/test_cuda_e3_evidence.py +++ b/tests/test_cuda_e3_evidence.py @@ -58,7 +58,7 @@ def _payload() -> dict[str, object]: "distribution": "rextio-tensorflow", "version": "0.1.2", "plugin_module": "rextio_tensorflow.plugin", - "native_module": "cuda_app._rextio_native", + "native_module": "_rextio_native", }, "source": { "core_commit": "7f47f0ce8cea0b6dbeb7fd3c733f65eeaa6bb5e0", @@ -193,6 +193,16 @@ def test_tampering_without_rehash_is_rejected() -> None: validate_envelope(envelope) +def test_native_module_identity_is_exactly_top_level_rextio_native() -> None: + assert _payload()["package"]["native_module"] == "_rextio_native" + envelope = _mutated( + ("payload", "package", "native_module"), + "cuda_app._rextio_native", + ) + with pytest.raises(EvidenceError, match="package or module identity"): + validate_envelope(envelope) + + @pytest.mark.parametrize( ("path", "value", "message"), ( From 1f08b84d04ebbba9c0803722e8d5f6428a249732 Mon Sep 17 00:00:00 2001 From: Rextio Date: Fri, 24 Jul 2026 14:32:48 +0900 Subject: [PATCH 07/15] test: bind CUDA E3 producer to evidence schema --- scripts/certify_cuda_candidate.py | 62 +++++++++++++++++------- tests/test_cuda_e3_manual_harness.py | 71 ++++++++++++++++++++++++++++ 2 files changed, 115 insertions(+), 18 deletions(-) diff --git a/scripts/certify_cuda_candidate.py b/scripts/certify_cuda_candidate.py index 0033e87..4d17743 100644 --- a/scripts/certify_cuda_candidate.py +++ b/scripts/certify_cuda_candidate.py @@ -51,6 +51,8 @@ "tensorflow_framework": "tensorflow/libtensorflow_framework.so.2", "pywrap_tensorflow_common": "tensorflow/python/lib_pywrap_tensorflow_common.so", } +ATOL = 1e-5 +RTOL = 1e-5 @dataclass(frozen=True) @@ -275,13 +277,23 @@ def execution_invariants(result: ExecutionResult) -> dict[str, Any]: """Translate observed execution facts into the verifier's closed shape.""" return { "execution": {"native_extension_executed": result.native_extension_executed, "kernel_activity_verified": False, "runtime_transfer_profiled": False, "runtime_provenance_checked": result.runtime_provenance_checked}, - "numerical": {"reference": "tensorflow-eager", "atol": 1e-5, "rtol": 1e-5, "max_scaled_error": result.max_scaled_error}, + "numerical": {"reference": "tensorflow-eager", "atol": ATOL, "rtol": RTOL, "max_scaled_error": result.max_scaled_error}, "output": {"device": "GPU:0", "dtype": "float32", "rank": 1, "shape": [4]}, "lifetime": {"inputs_unchanged": result.inputs_unchanged, "output_survives_input_gc": result.output_lifetime, "repeated_calls": result.repeated_calls}, "negative_boundary": {"cpu_input_rejected": result.cpu_input_rejected, "float64_rejected": result.float64_rejected, "wrong_rank_rejected": result.wrong_rank_rejected, "watched_tape_rejected": result.gradient_tape_rejected, "forward_accumulator_rejected": result.forward_accumulator_rejected}, } +def is_gpu0_device(device: str) -> bool: + """Recognize TensorFlow's canonical GPU:0 device-name suffix.""" + return device.endswith("/device:GPU:0") + + +def tolerance_scaled_error(difference: Any, reference: Any) -> Any: + """Return the approved absolute-plus-relative tolerance scaled error.""" + return abs(difference) / (ATOL + RTOL * abs(reference)) + + def _add_sources(*roots: Path) -> None: for root in reversed(roots): source = str(root / "src") @@ -375,18 +387,19 @@ def execute_e3(tf: Any, python_dir: Path) -> ExecutionResult: reference = tf.reduce_mean(tf.nn.relu(tf.nn.bias_add(tf.matmul(x, weight), bias)), axis=1) snapshots = tuple(tf.identity(value) for value in (x, weight, bias)) output = inference(x, weight, bias) - tf.debugging.assert_near(output, reference, rtol=1e-5, atol=1e-5) - denominator = tf.maximum(tf.abs(reference), tf.constant(1e-12, tf.float32)) - scaled = float(tf.reduce_max(tf.abs(output - reference) / denominator).numpy()) - if output.device.split(":")[0].endswith("GPU") is False or output.dtype != tf.float32 or output.shape != (4,): + tf.debugging.assert_near(output, reference, rtol=RTOL, atol=ATOL) + scaled = float(tf.reduce_max(tolerance_scaled_error(output - reference, reference)).numpy()) + if not is_gpu0_device(output.device) or output.dtype != tf.float32 or output.shape != (4,): raise RuntimeError("native output violated exact GPU:0 float32 rank-1 shape [4]") for original, snapshot in zip((x, weight, bias), snapshots, strict=True): tf.debugging.assert_equal(original, snapshot) + if not is_gpu0_device(original.device) or not is_gpu0_device(snapshot.device): + raise RuntimeError("native input or snapshot left exact TensorFlow GPU:0") del x, weight, bias gc.collect() - tf.debugging.assert_near(output, reference, rtol=1e-5, atol=1e-5) + tf.debugging.assert_near(output, reference, rtol=RTOL, atol=ATOL) for _ in range(3): - tf.debugging.assert_near(inference(*snapshots), reference, rtol=1e-5, atol=1e-5) + tf.debugging.assert_near(inference(*snapshots), reference, rtol=RTOL, atol=ATOL) with tf.device("/CPU:0"): cpu = tf.constant([[1., 2., 3.]], tf.float32) cases = (cpu, tf.cast(snapshots[0], tf.float64), tf.reshape(snapshots[0], (2, 2, 3))) @@ -419,6 +432,21 @@ def _artifact(role: str, label: str, path: Path) -> dict[str, Any]: return {"role": role, "label": label, "sha256": _sha256(path), "size_bytes": path.stat().st_size} +def build_payload(*, source: dict[str, Any], environment: dict[str, Any], toolchain: dict[str, str], artifacts: list[dict[str, Any]], runtime_images: list[dict[str, Any]], bindings: dict[str, Any], result: ExecutionResult) -> dict[str, Any]: + """Build precisely the payload accepted by the offline closed verifier.""" + return { + "contract": {"evidence_schema": "tensorflow-cuda-e3-real-nvidia-v1", "verification_scope": "schema-and-integrity-only", "producer_assertions": "self-attested-by-manual-harness", "support_claim": False, "certification_ready": False, "plugin_api": "1.6"}, + "package": {"distribution": "rextio-tensorflow", "version": "0.1.2", "plugin_module": "rextio_tensorflow.plugin", "native_module": "_rextio_native"}, + "source": source, + "environment": environment, + "toolchain": toolchain, + "artifacts": artifacts, + "runtime_images": runtime_images, + "orchestration": {"provider_id": PROVIDER_ID, "capability_id": CAPABILITY_ID, "device": "cuda:0", "input_residency": "device", "dtype": "float32", "ranks": [1, 2], "operations": ["tf.matmul", "tf.nn.bias_add", "tf.nn.relu", "tf.reduce_mean-axis1"], "artifact_profile_sha256": bindings["artifact_profile_sha256"], "authorization_sha256": bindings["authorization_sha256"], "provider_lock_sha256": bindings["lock_sha256"], "probe_sha256": bindings["probe_sha256"], "observations_sha256": bindings["observations_sha256"]}, + "invariants": execution_invariants(result), + } + + def main() -> int: """Collect one new self-attested evidence envelope in the closed schema.""" try: @@ -461,17 +489,15 @@ def main() -> int: _artifact("generated_cargo_lock", "generated/Cargo.lock", rust_dir / "Cargo.lock"), _artifact("native_extension", "python/_rextio_native" + sysconfig.get_config_var("EXT_SUFFIX"), extension), ] - payload = { - "contract": {"evidence_schema": "tensorflow-cuda-e3-real-nvidia-v1", "verification_scope": "schema-and-integrity-only", "producer_assertions": "self-attested-by-manual-harness", "support_claim": False, "certification_ready": False, "plugin_api": "1.6"}, - "package": {"distribution": "rextio-tensorflow", "version": "0.1.2", "plugin_module": "rextio_tensorflow.plugin", "native_module": "_rextio_native"}, - "source": {"core_commit": core.head, "core_clean": core.clean, "provider_commit": provider.head, "provider_clean": provider.clean, "plugin_commit": plugin.head, "plugin_clean": plugin.clean, "base_candidate_commit": BASE_CANDIDATE_COMMIT, "plugin_ancestry_checked": True}, - "environment": {"os": "Linux", "arch": "x86_64", "libc": "GNU", "python_implementation": "CPython", "python_version": "3.11", "tensorflow_version": tf.__version__, "cuda_driver_version": bindings["driver_version"], "gpu": {"ordinal": 0, "sm": args.sm}}, - "toolchain": toolchain, - "artifacts": artifacts, - "runtime_images": runtime_images, - "orchestration": {"provider_id": PROVIDER_ID, "capability_id": CAPABILITY_ID, "device": "cuda:0", "input_residency": "device", "dtype": "float32", "ranks": [1, 2], "operations": ["tf.matmul", "tf.nn.bias_add", "tf.nn.relu", "tf.reduce_mean-axis1"], "artifact_profile_sha256": bindings["artifact_profile_sha256"], "authorization_sha256": bindings["authorization_sha256"], "provider_lock_sha256": bindings["lock_sha256"], "probe_sha256": bindings["probe_sha256"], "observations_sha256": bindings["observations_sha256"]}, - "invariants": execution_invariants(result), - } + payload = build_payload( + source={"core_commit": core.head, "core_clean": core.clean, "provider_commit": provider.head, "provider_clean": provider.clean, "plugin_commit": plugin.head, "plugin_clean": plugin.clean, "base_candidate_commit": BASE_CANDIDATE_COMMIT, "plugin_ancestry_checked": True}, + environment={"os": "Linux", "arch": "x86_64", "libc": "GNU", "python_implementation": "CPython", "python_version": "3.11", "tensorflow_version": tf.__version__, "cuda_driver_version": bindings["driver_version"], "gpu": {"ordinal": 0, "sm": args.sm}}, + toolchain=toolchain, + artifacts=artifacts, + runtime_images=runtime_images, + bindings=bindings, + result=result, + ) envelope = verifier.make_envelope(payload) verifier.validate_envelope(envelope) atomic_create(args.output, verifier.canonical_bytes(envelope)) diff --git a/tests/test_cuda_e3_manual_harness.py b/tests/test_cuda_e3_manual_harness.py index 9bddae3..8dcf9ca 100644 --- a/tests/test_cuda_e3_manual_harness.py +++ b/tests/test_cuda_e3_manual_harness.py @@ -245,6 +245,77 @@ def test_execution_payload_records_only_observed_boundaries_and_no_profiler_clai assert "no_host_fallback_observed" not in encoded +def test_producer_payload_is_accepted_by_the_offline_verifier_without_tensorflow() -> None: + module = _module() + from scripts import verify_cuda_e3_evidence as verifier + + digest = "a" * 64 + result = module.ExecutionResult( + native_extension_executed=True, + numerical_parity=True, + max_scaled_error=0.25, + inputs_unchanged=True, + output_lifetime=True, + repeated_calls=True, + cpu_input_rejected=True, + float64_rejected=True, + wrong_rank_rejected=True, + gradient_tape_rejected=True, + forward_accumulator_rejected=True, + inputs_on_gpu=True, + output_on_gpu=True, + runtime_provenance_checked=True, + ) + payload = module.build_payload( + source={ + "core_commit": verifier.CORE_COMMIT, + "core_clean": True, + "provider_commit": verifier.PROVIDER_COMMIT, + "provider_clean": True, + "plugin_commit": "b" * 40, + "plugin_clean": True, + "base_candidate_commit": verifier.BASE_CANDIDATE_COMMIT, + "plugin_ancestry_checked": True, + }, + environment={ + "os": "Linux", + "arch": "x86_64", + "libc": "GNU", + "python_implementation": "CPython", + "python_version": "3.11", + "tensorflow_version": "2.21.0", + "cuda_driver_version": 12080, + "gpu": {"ordinal": 0, "sm": "sm_80"}, + }, + toolchain={"rustc_version": "1.93.1", "cargo_version": "1.93.1", "target": module.TARGET}, + artifacts=[ + {"role": role, "label": f"evidence/{role}", "sha256": digest, "size_bytes": 1} + for role in sorted(verifier.ARTIFACT_ROLES) + ], + runtime_images=[ + {"role": role, "wheel_path": path, "sha256": digest, "size_bytes": 1, "build_id": None, "mapped": True} + for role, path in verifier.RUNTIME_IMAGES.items() + ], + bindings={ + "artifact_profile_sha256": digest, + "authorization_sha256": digest, + "lock_sha256": digest, + "probe_sha256": digest, + "observations_sha256": digest, + }, + result=result, + ) + assert payload["package"]["native_module"] == "_rextio_native" + assert verifier.validate_envelope(verifier.make_envelope(payload)) == payload + + +def test_gpu_device_and_tolerance_helpers_are_exact_and_gpu_free() -> None: + module = _module() + assert module.is_gpu0_device("/job:localhost/replica:0/task:0/device:GPU:0") + assert not module.is_gpu0_device("/job:localhost/replica:0/task:0/device:GPU:1") + assert module.tolerance_scaled_error(2e-5, 1.0) == pytest.approx(1.0) + + def test_source_contains_explicit_tensorflow_before_extension_and_provenance_guards() -> None: source = SCRIPT.read_text(encoding="utf-8") assert source.index("import tensorflow as tf") < source.index("import_generated_inference(") From 3cf8215aa3ac72c709183cc1cbb443e98c7ff6ce Mon Sep 17 00:00:00 2001 From: Rextio Date: Fri, 24 Jul 2026 14:34:34 +0900 Subject: [PATCH 08/15] Require TensorFlow runtime build IDs --- scripts/verify_cuda_e3_evidence.py | 11 +++++------ tests/test_cuda_e3_evidence.py | 10 ++++++++-- 2 files changed, 13 insertions(+), 8 deletions(-) diff --git a/scripts/verify_cuda_e3_evidence.py b/scripts/verify_cuda_e3_evidence.py index 8e0838d..f80ba46 100644 --- a/scripts/verify_cuda_e3_evidence.py +++ b/scripts/verify_cuda_e3_evidence.py @@ -398,12 +398,11 @@ def _validate_artifacts(payload: dict[str, Any]) -> None: f"payload.runtime_images[{index}].size_bytes", ) build_id = image["build_id"] - if build_id is not None: - _string( - build_id, - f"payload.runtime_images[{index}].build_id", - pattern=BUILD_ID, - ) + _string( + build_id, + f"payload.runtime_images[{index}].build_id", + pattern=BUILD_ID, + ) if image["mapped"] is not True: raise EvidenceError(f"runtime image {role} must be self-attested mapped=true") if image_roles != set(RUNTIME_IMAGES): diff --git a/tests/test_cuda_e3_evidence.py b/tests/test_cuda_e3_evidence.py index 0a7b7cc..ad0e2e6 100644 --- a/tests/test_cuda_e3_evidence.py +++ b/tests/test_cuda_e3_evidence.py @@ -100,7 +100,7 @@ def _payload() -> dict[str, object]: "wheel_path": wheel_path, "sha256": HASH, "size_bytes": 1, - "build_id": None, + "build_id": "b" * 40, "mapped": True, } for role, wheel_path in runtime_rows @@ -284,7 +284,7 @@ def test_recursively_rejects_path_url_and_credential_leaks(leak: str) -> None: validate_envelope(envelope) -def test_runtime_images_are_exact_and_build_id_is_nullable_bounded() -> None: +def test_runtime_images_are_exact_and_build_id_is_required_lowercase_bounded() -> None: envelope = _mutated( ("payload", "runtime_images", 0, "wheel_path"), "tensorflow/libtensorflow_cc.so", @@ -292,6 +292,12 @@ def test_runtime_images_are_exact_and_build_id_is_nullable_bounded() -> None: with pytest.raises(EvidenceError, match="wrong wheel-relative"): validate_envelope(envelope) envelope = _mutated(("payload", "runtime_images", 0, "build_id"), "abc") + with pytest.raises(EvidenceError, match="invalid format"): + validate_envelope(envelope) + envelope = _mutated(("payload", "runtime_images", 0, "build_id"), None) + with pytest.raises(EvidenceError, match="non-empty string"): + validate_envelope(envelope) + envelope = _mutated(("payload", "runtime_images", 0, "build_id"), "B" * 40) with pytest.raises(EvidenceError, match="invalid format"): validate_envelope(envelope) envelope = _mutated(("payload", "runtime_images", 0, "mapped"), False) From 42a1281961d17ca5e40ae431fbab62f3a81074ff Mon Sep 17 00:00:00 2001 From: Rextio Date: Fri, 24 Jul 2026 14:34:45 +0900 Subject: [PATCH 09/15] fix: tighten CUDA E3 evidence capture --- scripts/certify_cuda_candidate.py | 38 ++++++++++++------ tests/test_cuda_e3_manual_harness.py | 59 ++++++++++++++++++++-------- 2 files changed, 69 insertions(+), 28 deletions(-) diff --git a/scripts/certify_cuda_candidate.py b/scripts/certify_cuda_candidate.py index 4d17743..6248123 100644 --- a/scripts/certify_cuda_candidate.py +++ b/scripts/certify_cuda_candidate.py @@ -127,6 +127,14 @@ def validate_requested_paths_and_values(args: argparse.Namespace, allowed_sms: s raise RuntimeError("--sm must be one of the verifier allowed architectures") +def validate_destinations_are_outside_checkouts(args: argparse.Namespace, roots: tuple[Path, Path, Path]) -> None: + """Keep generated state outside every checkout whose cleanliness is attested.""" + destinations = (args.work_dir.resolve(), args.output.resolve()) + for destination in destinations: + if any(destination.is_relative_to(root) for root in roots): + raise RuntimeError("work-dir and output must be outside all source checkouts") + + def atomic_create(output: Path, data: bytes) -> None: """Create, never replace, canonical evidence; clean a failed temporary.""" descriptor, temporary_name = tempfile.mkstemp(prefix=f".{output.name}.", dir=output.parent) @@ -199,11 +207,14 @@ def validate_and_bind_provider_plan(plan: dict[str, Any], sm: str, probe_sha256: authorization = plan["lowering_authorization"] lock = plan["lock"] profile = plan.get("artifact_profile", {}) + preflight = plan["preflight"] report = plan["report"] if profile.get("target_triple") != TARGET: raise RuntimeError("provider artifact profile target changed") if report.get("support_claim") is not False or report.get("certification_tier") != "build-only": raise RuntimeError("provider support/certification claim is not build-only") + if report.get("status") != "ready" or report.get("reason_codes") != []: + raise RuntimeError("provider preflight is not unqualified ready") expected = { "provider_id": PROVIDER_ID, "capability_id": CAPABILITY_ID, @@ -213,10 +224,12 @@ def validate_and_bind_provider_plan(plan: dict[str, Any], sm: str, probe_sha256: if any(authorization.get(key) != value for key, value in expected.items()): raise RuntimeError("provider authorization changed") profile_hash = authorization.get("artifact_profile_sha256") - if not isinstance(profile_hash, str) or not HEX64.fullmatch(profile_hash) or lock.get("artifact_profile_sha256") != profile_hash: + if not isinstance(profile_hash, str) or not HEX64.fullmatch(profile_hash) or lock.get("artifact_profile_sha256") != profile_hash or _canonical_hash(profile) != profile_hash: raise RuntimeError("provider profile authorization is unbound") - if not isinstance(lock.get("preflight_sha256"), str) or not HEX64.fullmatch(lock["preflight_sha256"]): + if not isinstance(lock.get("preflight_sha256"), str) or not HEX64.fullmatch(lock["preflight_sha256"]) or _canonical_hash(preflight) != lock["preflight_sha256"]: raise RuntimeError("provider lock is invalid") + if not isinstance(probe_sha256, str) or not HEX64.fullmatch(probe_sha256): + raise RuntimeError("provider probe hash is invalid") observations = report.get("observations") if not isinstance(observations, list): raise RuntimeError("provider observations are absent") @@ -254,22 +267,16 @@ def capture_runtime_images(wheel_root: Path, maps: str, read_build_id: Callable[ "tensorflow_framework": wheel_root / "tensorflow" / "libtensorflow_framework.so.2", "pywrap_tensorflow_common": wheel_root / "tensorflow" / "python" / "lib_pywrap_tensorflow_common.so", } - # The legacy spelling is accepted only to keep the GPU-free unit contract - # focused on map parsing; a real production payload always uses canonical. - legacy = wheel_root / "python" / "_pywrap_tensorflow_internal.so" - if not canonical["pywrap_tensorflow_common"].is_file() and legacy.is_file(): - canonical = { - "tensorflow_cc": wheel_root / "libtensorflow_cc.so.2", - "tensorflow_framework": wheel_root / "libtensorflow_framework.so.2", - "tensorflow_pywrap": legacy, - } rows: list[dict[str, Any]] = [] for role, path in canonical.items(): resolved = path.resolve() if not path.is_file() or str(resolved) not in maps: raise RuntimeError(f"expected TensorFlow runtime image is not mapped: {role}") relative = path.relative_to(wheel_root).as_posix() - rows.append({"role": role, "wheel_path": relative, "sha256": _sha256(path), "size_bytes": path.stat().st_size, "build_id": read_build_id(path), "mapped": True}) + build_id = read_build_id(path) + if not build_id: + raise RuntimeError(f"expected TensorFlow runtime image has no build ID: {role}") + rows.append({"role": role, "wheel_path": relative, "sha256": _sha256(path), "size_bytes": path.stat().st_size, "build_id": build_id, "mapped": True}) return rows @@ -458,6 +465,7 @@ def main() -> int: toolchain = validate_host() roots = tuple(path.resolve() for path in (args.tensorflow_root, args.core_root, args.provider_root)) tensorflow_root, core_root, provider_root = roots + validate_destinations_are_outside_checkouts(args, roots) core = checkout_identity(core_root) provider = checkout_identity(provider_root) plugin = checkout_identity(tensorflow_root) @@ -489,6 +497,12 @@ def main() -> int: _artifact("generated_cargo_lock", "generated/Cargo.lock", rust_dir / "Cargo.lock"), _artifact("native_extension", "python/_rextio_native" + sysconfig.get_config_var("EXT_SUFFIX"), extension), ] + core = checkout_identity(core_root) + provider = checkout_identity(provider_root) + plugin = checkout_identity(tensorflow_root) + validate_checkout(core, CORE_COMMIT, CORE_COMMIT) + validate_checkout(provider, PROVIDER_COMMIT, PROVIDER_COMMIT) + validate_checkout(plugin, args.expected_tensorflow_commit, BASE_CANDIDATE_COMMIT) payload = build_payload( source={"core_commit": core.head, "core_clean": core.clean, "provider_commit": provider.head, "provider_clean": provider.clean, "plugin_commit": plugin.head, "plugin_clean": plugin.clean, "base_candidate_commit": BASE_CANDIDATE_COMMIT, "plugin_ancestry_checked": True}, environment={"os": "Linux", "arch": "x86_64", "libc": "GNU", "python_implementation": "CPython", "python_version": "3.11", "tensorflow_version": tf.__version__, "cuda_driver_version": bindings["driver_version"], "gpu": {"ordinal": 0, "sm": args.sm}}, diff --git a/tests/test_cuda_e3_manual_harness.py b/tests/test_cuda_e3_manual_harness.py index 8dcf9ca..045d653 100644 --- a/tests/test_cuda_e3_manual_harness.py +++ b/tests/test_cuda_e3_manual_harness.py @@ -81,6 +81,18 @@ def test_cli_requires_exclusive_work_and_output_paths(tmp_path: Path) -> None: module.validate_requested_paths_and_values(args, {"sm_80"}) +def test_destinations_must_be_outside_every_cleanliness_attested_checkout(tmp_path: Path) -> None: + module = _module() + roots = tuple(tmp_path / name for name in ("tensorflow", "core", "provider")) + for root in roots: + root.mkdir() + args = argparse.Namespace(work_dir=tmp_path / "work", output=tmp_path / "evidence.json") + module.validate_destinations_are_outside_checkouts(args, roots) + args.output = roots[0] / "evidence.json" + with pytest.raises(RuntimeError, match="outside"): + module.validate_destinations_are_outside_checkouts(args, roots) + + def test_canonical_atomic_create_is_exclusive_and_cleans_temporary( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: @@ -136,12 +148,27 @@ def fake_run(command, **kwargs): def test_provider_observations_are_validated_and_bound_to_hashes() -> None: module = _module() - profile_hash = "1" * 64 + profile = {"target_triple": module.TARGET} + preflight = { + "provider_id": module.PROVIDER_ID, + "status": "ready", + "reason_codes": [], + "observations": [ + {"key": "driver.version", "value": "12080"}, + {"key": "selected.device", "value": "0"}, + {"key": "selected.sm", "value": "sm_80"}, + {"key": "probe.schema", "value": "1"}, + {"key": "framework.runtime", "value": "tensorflow-tfe"}, + ], + "support_claim": False, + } + profile_hash = module._canonical_hash(profile) plan = { - "artifact_profile": {"target_triple": module.TARGET}, + "artifact_profile": profile, + "preflight": preflight, "lock": { "artifact_profile_sha256": profile_hash, - "preflight_sha256": "2" * 64, + "preflight_sha256": module._canonical_hash(preflight), }, "lowering_authorization": { "provider_id": module.PROVIDER_ID, @@ -155,13 +182,7 @@ def test_provider_observations_are_validated_and_bound_to_hashes() -> None: "support_claim": False, "certification_tier": "build-only", "reason_codes": [], - "observations": [ - {"key": "driver.version", "value": "12080"}, - {"key": "selected.device", "value": "0"}, - {"key": "selected.sm", "value": "sm_80"}, - {"key": "probe.schema", "value": "1"}, - {"key": "framework.runtime", "value": "tensorflow-tfe"}, - ], + "observations": preflight["observations"], }, } result = module.validate_and_bind_provider_plan(plan, "sm_80", "3" * 64) @@ -178,14 +199,18 @@ def test_provider_observations_are_validated_and_bound_to_hashes() -> None: plan["report"]["support_claim"] = True with pytest.raises(RuntimeError, match="support"): module.validate_and_bind_provider_plan(plan, "sm_80", "3" * 64) + plan["report"]["support_claim"] = False + plan["artifact_profile"]["target_triple"] = "forged-target" + with pytest.raises(RuntimeError, match="profile"): + module.validate_and_bind_provider_plan(plan, "sm_80", "3" * 64) def test_runtime_dso_capture_requires_expected_mapped_wheel_images(tmp_path: Path) -> None: module = _module() - wheel = tmp_path / "tensorflow" - pywrap = wheel / "python" / "_pywrap_tensorflow_internal.so" - cc = wheel / "libtensorflow_cc.so.2" - framework = wheel / "libtensorflow_framework.so.2" + wheel = tmp_path + pywrap = wheel / "tensorflow" / "python" / "lib_pywrap_tensorflow_common.so" + cc = wheel / "tensorflow" / "libtensorflow_cc.so.2" + framework = wheel / "tensorflow" / "libtensorflow_framework.so.2" for path in (pywrap, cc, framework): path.parent.mkdir(parents=True, exist_ok=True) path.write_bytes(path.name.encode()) @@ -196,7 +221,7 @@ def test_runtime_dso_capture_requires_expected_mapped_wheel_images(tmp_path: Pat read_build_id=lambda path: f"build-{path.name}", ) assert {row["role"] for row in identities} == { - "tensorflow_pywrap", + "pywrap_tensorflow_common", "tensorflow_cc", "tensorflow_framework", } @@ -204,6 +229,8 @@ def test_runtime_dso_capture_requires_expected_mapped_wheel_images(tmp_path: Pat assert all(not row["wheel_path"].startswith("/") for row in identities) with pytest.raises(RuntimeError, match="mapped"): module.capture_runtime_images(wheel, maps.replace(str(cc), ""), read_build_id=lambda _: "x") + with pytest.raises(RuntimeError, match="build ID"): + module.capture_runtime_images(wheel, maps, read_build_id=lambda _: None) def test_execution_payload_records_only_observed_boundaries_and_no_profiler_claims() -> None: @@ -293,7 +320,7 @@ def test_producer_payload_is_accepted_by_the_offline_verifier_without_tensorflow for role in sorted(verifier.ARTIFACT_ROLES) ], runtime_images=[ - {"role": role, "wheel_path": path, "sha256": digest, "size_bytes": 1, "build_id": None, "mapped": True} + {"role": role, "wheel_path": path, "sha256": digest, "size_bytes": 1, "build_id": "b" * 8, "mapped": True} for role, path in verifier.RUNTIME_IMAGES.items() ], bindings={ From 2015c5ee6c1c8a85ca30f92eaddc89ac4ff6086d Mon Sep 17 00:00:00 2001 From: Rextio Date: Fri, 24 Jul 2026 14:41:00 +0900 Subject: [PATCH 10/15] docs: align CUDA E3 evidence operation contract --- .github/workflows/ci.yml | 19 ++++++++ README.md | 10 +++-- docs/cuda-build-only-0.1.2.md | 85 ++++++++++++++++++++++++----------- 3 files changed, 84 insertions(+), 30 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index d1ae4d6..54d5fe4 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -303,6 +303,25 @@ jobs: run: | python scripts/certify_cuda_candidate.py --help python scripts/verify_cuda_e3_evidence.py --help + - name: Prove both evidence scripts import without TensorFlow + run: | + python - <<'PY' + import importlib.util + import sys + from pathlib import Path + + for name in ("certify_cuda_candidate", "verify_cuda_e3_evidence"): + path = Path("scripts") / f"{name}.py" + spec = importlib.util.spec_from_file_location(name, path) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + sys.modules[name] = module + spec.loader.exec_module(module) + assert not any( + name == "tensorflow" or name.startswith("tensorflow.") + for name in sys.modules + ) + PY ci-gate: name: ci-gate diff --git a/README.md b/README.md index 508b73e..934c99a 100644 --- a/README.md +++ b/README.md @@ -33,10 +33,12 @@ requires exact authorization from The hosted CUDA job is compile/link-only: it never installs/imports TensorFlow, loads the extension, or executes CUDA. A separately opt-in real-NVIDIA -first-stage harness can record execution/parity/lifetime evidence, but it -still records `native_extension_executed=true`, -`kernel_activity_verified=false`, and `runtime_transfer_profiled=false`. -That evidence is not kernel/profile certification or CUDA support. See +first-stage harness may produce self-attested execution/parity/lifetime +evidence; its closed schema requires `kernel_activity_verified=false` and +`runtime_transfer_profiled=false`, and preserves `support_claim=false` and +`certification_ready=false`. The offline verifier establishes schema and +payload integrity only, not GPU execution, hardware certification, or CUDA +support. See [the CUDA build-only and manual-evidence contract](docs/cuda-build-only-0.1.2.md) for the exact Linux GNU/CPython 3.11/TF 2.21.0/Rust 1.93.1 pins, clean candidate checkout, GPU:0/permitted-SM boundary, and commands. diff --git a/docs/cuda-build-only-0.1.2.md b/docs/cuda-build-only-0.1.2.md index ae2b188..eb8dfdc 100644 --- a/docs/cuda-build-only-0.1.2.md +++ b/docs/cuda-build-only-0.1.2.md @@ -102,26 +102,33 @@ This is a frozen environment, not a portability recipe: - CPython 3.11, TensorFlow `2.21.0`, and Rust `1.93.1`. - Core checkout exactly `7f47f0ce8cea0b6dbeb7fd3c733f65eeaa6bb5e0` and CUDA provider checkout exactly `cf65733f06b91a801f9806367f09948ee7162540`. -- A clean TensorFlow-plugin checkout at candidate commit exactly - `16e368a2e73be58d4cc51da1672a8a842e394fbd`; pass that value explicitly via - `--expected-tensorflow-commit`. +- A clean TensorFlow-plugin checkout of the unreleased `0.1.2` integration + branch. Record its full current SHA with `git rev-parse HEAD` and pass that + exact, lowercase 40-character value explicitly via + `--expected-tensorflow-commit`; the harness verifies that it descends from + the frozen E3 base. - Exactly one usable `GPU:0`, with a permitted architecture from this closed set: `sm_60`, `sm_61`, `sm_70`, `sm_72`, `sm_75`, `sm_80`, `sm_86`, `sm_87`, `sm_89`, or `sm_90`. Other ordinals and SM values are rejected rather than generalized. +- GNU binutils, including `readelf`, on `PATH`. The harness records GNU build + IDs from the TensorFlow wheel images and fails closed when an expected image + has no build ID. The harness deliberately has no `toolkit_root` setting or command-line option. It reuses the active TensorFlow wheel and its already-loaded images; pointing at an independent CUDA toolkit would violate the runtime-reuse contract. -Use independent checkout and output directories so neither evidence nor build -products can be confused with a source checkout: +Use independent checkout, output, and work directories. The output file and +the new exclusive work directory must be outside **all three** clean source +checkouts; the harness rejects paths inside any attested checkout. Do not +create the work directory itself: the harness requires it not to exist yet. ```bash export E3_ROOT="$HOME/rextio-tf-e3-manual-$(date +%Y%m%d-%H%M%S)" export E3_OUT="$E3_ROOT/evidence-output" export E3_BUILD="$E3_ROOT/isolated-build" -mkdir -p "$E3_ROOT/checkouts" "$E3_OUT" "$E3_BUILD" +mkdir -p "$E3_ROOT/checkouts" "$E3_OUT" git clone https://github.com/rextio/rextio.git "$E3_ROOT/checkouts/rextio" git -C "$E3_ROOT/checkouts/rextio" checkout --detach \ @@ -130,10 +137,12 @@ git clone https://github.com/rextio/rextio-device-cuda.git \ "$E3_ROOT/checkouts/rextio-device-cuda" git -C "$E3_ROOT/checkouts/rextio-device-cuda" checkout --detach \ cf65733f06b91a801f9806367f09948ee7162540 -git clone https://github.com/rextio/rextio-tensorflow.git \ +git clone --branch 0.1.2 --single-branch \ + https://github.com/rextio/rextio-tensorflow.git \ "$E3_ROOT/checkouts/rextio-tensorflow" -git -C "$E3_ROOT/checkouts/rextio-tensorflow" checkout --detach \ - 16e368a2e73be58d4cc51da1672a8a842e394fbd +export TF_ROOT="$E3_ROOT/checkouts/rextio-tensorflow" +export TF_COMMIT="$(git -C "$TF_ROOT" rev-parse HEAD)" +test "${#TF_COMMIT}" -eq 40 python3.11 -m venv "$E3_ROOT/venv" "$E3_ROOT/venv/bin/python" -m pip install --upgrade pip @@ -142,28 +151,52 @@ python3.11 -m venv "$E3_ROOT/venv" "$E3_ROOT/checkouts/rextio" "$E3_ROOT/checkouts/rextio-device-cuda" \ "$E3_ROOT/checkouts/rextio-tensorflow" rustup toolchain install 1.93.1 --profile minimal +command -v readelf +readelf --version ``` -Import TensorFlow before invoking the candidate. This is required to establish -the wheel-image reuse boundary, rather than an optional smoke test: +Import TensorFlow **in the same process that invokes the harness**. This is +required to establish the wheel-image reuse boundary, rather than an optional +smoke test. Set `TF_SM` to the actual permitted architecture of the sole usable +`GPU:0` (the example uses `sm_80` only as a placeholder): ```bash -cd "$E3_ROOT/checkouts/rextio-tensorflow" -"$E3_ROOT/venv/bin/python" -c 'import tensorflow as tf; assert tf.__version__ == "2.21.0"' -"$E3_ROOT/venv/bin/python" scripts/certify_cuda_candidate.py \ - --output "$E3_OUT/cuda-e3-first-stage.json" \ - --work-dir "$E3_BUILD" \ - --core-root "$E3_ROOT/checkouts/rextio" \ - --provider-root "$E3_ROOT/checkouts/rextio-device-cuda" \ - --expected-tensorflow-commit 16e368a2e73be58d4cc51da1672a8a842e394fbd \ - --sm sm_80 +cd "$TF_ROOT" +export TF_SM=sm_80 +E3_OUTPUT="$E3_OUT/cuda-e3-first-stage.json" E3_WORK="$E3_BUILD" \ +E3_CORE="$E3_ROOT/checkouts/rextio" \ +E3_PROVIDER="$E3_ROOT/checkouts/rextio-device-cuda" \ +"$E3_ROOT/venv/bin/python" - <<'PY' +import os +import sys + +import tensorflow as tf + +assert tf.__version__ == "2.21.0" +from scripts import certify_cuda_candidate + +sys.argv = [ + "certify_cuda_candidate.py", + "--output", os.environ["E3_OUTPUT"], + "--work-dir", os.environ["E3_WORK"], + "--core-root", os.environ["E3_CORE"], + "--provider-root", os.environ["E3_PROVIDER"], + "--expected-tensorflow-commit", os.environ["TF_COMMIT"], + "--sm", os.environ["TF_SM"], +] +raise SystemExit(certify_cuda_candidate.main()) +PY "$E3_ROOT/venv/bin/python" scripts/verify_cuda_e3_evidence.py \ "$E3_OUT/cuda-e3-first-stage.json" ``` -The evidence records `native_extension_executed=true`, but intentionally -records `kernel_activity_verified=false` and `runtime_transfer_profiled=false`. -Accordingly it is execution, numerical-parity, and borrowed-object-lifetime -evidence only. It is **not** kernel-activity certification, a transfer/profile -measurement, CUDA support, or a performance claim. A successful harness and -verifier run leave `support_claim=false` and `certification_ready=false`. +The producer self-attests `native_extension_executed=true` only if the bounded +harness reaches that observation. It intentionally records +`kernel_activity_verified=false` and `runtime_transfer_profiled=false`. +The offline verifier checks canonical schema and payload integrity only; it +does not authenticate the producer, prove execution, recompute artifact +hashes, certify hardware, or confer CUDA support. Any evidence remains +self-attested execution, numerical-parity, and borrowed-object-lifetime +evidence, never a GPU-success claim, kernel-activity certification, +transfer/profile measurement, CUDA support, or a performance claim. It always +leaves `support_claim=false` and `certification_ready=false`. From 64b6e42ce30dfff3a8a6e7b44ae589157ca0d82a Mon Sep 17 00:00:00 2001 From: Rextio Date: Fri, 24 Jul 2026 14:46:19 +0900 Subject: [PATCH 11/15] docs: clarify CUDA E3 pre-merge checkout --- docs/cuda-build-only-0.1.2.md | 21 +++++++++++++++------ 1 file changed, 15 insertions(+), 6 deletions(-) diff --git a/docs/cuda-build-only-0.1.2.md b/docs/cuda-build-only-0.1.2.md index eb8dfdc..2ad30f2 100644 --- a/docs/cuda-build-only-0.1.2.md +++ b/docs/cuda-build-only-0.1.2.md @@ -102,9 +102,13 @@ This is a frozen environment, not a portability recipe: - CPython 3.11, TensorFlow `2.21.0`, and Rust `1.93.1`. - Core checkout exactly `7f47f0ce8cea0b6dbeb7fd3c733f65eeaa6bb5e0` and CUDA provider checkout exactly `cf65733f06b91a801f9806367f09948ee7162540`. -- A clean TensorFlow-plugin checkout of the unreleased `0.1.2` integration - branch. Record its full current SHA with `git rev-parse HEAD` and pass that - exact, lowercase 40-character value explicitly via +- A clean TensorFlow-plugin checkout selected by `TF_REF`. After this PR is + integrated, set `TF_REF=0.1.2`; that is the default runnable path. Until + then, the current `0.1.2` target branch does not contain these two scripts, + so a reviewer/operator must set `TF_REF` to the current immutable full + PR-head SHA instead. Do not use a moving feature-branch name or record that + self-referential SHA in this document. After checkout, derive the full + lowercase SHA with `git rev-parse HEAD` and pass it explicitly via `--expected-tensorflow-commit`; the harness verifies that it descends from the frozen E3 base. - Exactly one usable `GPU:0`, with a permitted architecture from this closed @@ -137,10 +141,15 @@ git clone https://github.com/rextio/rextio-device-cuda.git \ "$E3_ROOT/checkouts/rextio-device-cuda" git -C "$E3_ROOT/checkouts/rextio-device-cuda" checkout --detach \ cf65733f06b91a801f9806367f09948ee7162540 -git clone --branch 0.1.2 --single-branch \ - https://github.com/rextio/rextio-tensorflow.git \ - "$E3_ROOT/checkouts/rextio-tensorflow" export TF_ROOT="$E3_ROOT/checkouts/rextio-tensorflow" +export TF_REF=0.1.2 +# Before this PR merges, replace 0.1.2 above with the current immutable full +# PR-head SHA. The 0.1.2 default becomes runnable only after integration. +git clone https://github.com/rextio/rextio-tensorflow.git "$TF_ROOT" +git -C "$TF_ROOT" fetch --tags origin "$TF_REF" +git -C "$TF_ROOT" checkout --detach "$TF_REF" +test -f "$TF_ROOT/scripts/certify_cuda_candidate.py" +test -f "$TF_ROOT/scripts/verify_cuda_e3_evidence.py" export TF_COMMIT="$(git -C "$TF_ROOT" rev-parse HEAD)" test "${#TF_COMMIT}" -eq 40 From bedbe6d7ef91c26991d36376be10753d1e544d63 Mon Sep 17 00:00:00 2001 From: Rextio Date: Fri, 24 Jul 2026 14:46:27 +0900 Subject: [PATCH 12/15] Harden CUDA E3 evidence verifier input handling --- scripts/verify_cuda_e3_evidence.py | 12 +++++++--- tests/test_cuda_e3_evidence.py | 38 +++++++++++++++++++++++++++++- 2 files changed, 46 insertions(+), 4 deletions(-) diff --git a/scripts/verify_cuda_e3_evidence.py b/scripts/verify_cuda_e3_evidence.py index f80ba46..26f84ad 100644 --- a/scripts/verify_cuda_e3_evidence.py +++ b/scripts/verify_cuda_e3_evidence.py @@ -596,17 +596,23 @@ def validate_document(raw: bytes) -> dict[str, Any]: EvidenceError(f"non-finite JSON value {value}") ), ) - except (UnicodeDecodeError, json.JSONDecodeError) as exc: + except (UnicodeDecodeError, json.JSONDecodeError, RecursionError) as exc: raise EvidenceError(f"malformed evidence JSON: {exc}") from exc try: expected = canonical_bytes(envelope) - except (TypeError, ValueError) as exc: + except (TypeError, ValueError, RecursionError) as exc: raise EvidenceError(f"evidence cannot be canonicalized: {exc}") from exc if raw != expected: raise EvidenceError("evidence bytes are not canonical JSON with exactly one final newline") return validate_envelope(envelope) +def read_document(path: Path) -> bytes: + """Read no more than the verifier's accepted document size plus one byte.""" + with path.open("rb") as handle: + return handle.read(MAX_BYTES + 1) + + def main(argv: list[str] | None = None) -> int: """Run the offline schema and integrity verifier CLI.""" parser = argparse.ArgumentParser(description=__doc__) @@ -617,7 +623,7 @@ def main(argv: list[str] | None = None) -> int: ) args = parser.parse_args(argv) try: - payload = validate_document(args.evidence.read_bytes()) + payload = validate_document(read_document(args.evidence)) except (OSError, EvidenceError) as exc: print(f"evidence schema/integrity verification failed: {exc}", file=sys.stderr) return 1 diff --git a/tests/test_cuda_e3_evidence.py b/tests/test_cuda_e3_evidence.py index ad0e2e6..7774d92 100644 --- a/tests/test_cuda_e3_evidence.py +++ b/tests/test_cuda_e3_evidence.py @@ -13,6 +13,7 @@ from scripts.verify_cuda_e3_evidence import ( BASE_CANDIDATE_COMMIT, EvidenceError, + MAX_BYTES, canonical_bytes, canonical_json, make_envelope, @@ -344,7 +345,7 @@ def test_malformed_nonfinite_oversize_and_depth_are_rejected(tmp_path: Path) -> with pytest.raises(EvidenceError): validate_document(raw) with pytest.raises(EvidenceError, match="maximum size"): - validate_document(b" " * 131_073) + validate_document(b" " * (MAX_BYTES + 1)) value: object = {} cursor = value for _ in range(14): @@ -366,6 +367,41 @@ def test_malformed_nonfinite_oversize_and_depth_are_rejected(tmp_path: Path) -> assert "schema/integrity verification failed" in completed.stderr +def test_cli_rejects_oversized_input_without_unbounded_read(tmp_path: Path) -> None: + oversized = tmp_path / "oversized.json" + oversized.write_bytes(b" " * (MAX_BYTES + 1)) + + completed = subprocess.run( + [sys.executable, str(SCRIPT), str(oversized)], + capture_output=True, + text=True, + check=False, + ) + + assert completed.returncode == 1 + assert "evidence exceeds maximum size" in completed.stderr + assert "Traceback" not in completed.stderr + + +def test_deep_json_recursion_is_a_controlled_verifier_failure(tmp_path: Path) -> None: + deeply_nested = b"[" * 20_000 + b"0" + b"]" * 20_000 + b"\n" + with pytest.raises(EvidenceError): + validate_document(deeply_nested) + + evidence = tmp_path / "deep.json" + evidence.write_bytes(deeply_nested) + completed = subprocess.run( + [sys.executable, str(SCRIPT), str(evidence)], + capture_output=True, + text=True, + check=False, + ) + + assert completed.returncode == 1 + assert "schema/integrity verification failed" in completed.stderr + assert "Traceback" not in completed.stderr + + def test_cli_help_names_schema_and_integrity_scope() -> None: completed = subprocess.run( [sys.executable, str(SCRIPT), "--help"], From e7b63af9f1d41943c57d72e8dbd415d1165c26f8 Mon Sep 17 00:00:00 2001 From: Rextio Date: Fri, 24 Jul 2026 14:49:33 +0900 Subject: [PATCH 13/15] fix(cuda): harden E3 manual harness integration --- scripts/certify_cuda_candidate.py | 91 +++++++++++++++++++++----- tests/test_cuda_e3_manual_harness.py | 97 +++++++++++++++++++++++++++- 2 files changed, 170 insertions(+), 18 deletions(-) diff --git a/scripts/certify_cuda_candidate.py b/scripts/certify_cuda_candidate.py index 6248123..e0d58da 100644 --- a/scripts/certify_cuda_candidate.py +++ b/scripts/certify_cuda_candidate.py @@ -11,6 +11,9 @@ import argparse import gc import hashlib +import importlib +import importlib.util +import inspect import json import os import platform @@ -188,7 +191,16 @@ def assert_frozen_source_contract(rust: str) -> None: def build_generated_extension(rust_dir: Path, python_dir: Path, environment: dict[str, str]) -> Path: """Build locked with Rust 1.93.1 and install the exact CPython suffix.""" - command = ["cargo", "+1.93.1", "build", "--locked", "--release", "--manifest-path", str(rust_dir / "Cargo.toml")] + manifest = rust_dir / "Cargo.toml" + lockfile = rust_dir / "Cargo.lock" + subprocess.run( + ["cargo", "+1.93.1", "generate-lockfile", "--manifest-path", str(manifest)], + check=True, + env=environment, + ) + if not lockfile.is_file() or lockfile.stat().st_size == 0: + raise RuntimeError("Cargo.lock was not created by pinned lockfile generation") + command = ["cargo", "+1.93.1", "build", "--locked", "--release", "--manifest-path", str(manifest)] subprocess.run(command, check=True, env=environment) candidates = tuple((rust_dir / "target" / "release").glob("*rextio_native*.so")) if len(candidates) != 1 or not candidates[0].is_file() or candidates[0].stat().st_size == 0: @@ -236,9 +248,12 @@ def validate_and_bind_provider_plan(plan: dict[str, Any], sm: str, probe_sha256: observed = {row.get("key"): row.get("value") for row in observations if isinstance(row, dict)} if observed.get("selected.device") != "0" or observed.get("selected.sm") != sm or observed.get("probe.schema") != "1" or observed.get("framework.runtime") != "tensorflow-tfe": raise RuntimeError("provider observations do not bind the selected CUDA E3 capability") + driver_text = observed.get("driver.version") + if not isinstance(driver_text, str): + raise RuntimeError("provider driver observation is invalid") try: - driver = int(observed["driver.version"]) - except (KeyError, TypeError, ValueError) as error: + driver = int(driver_text) + except ValueError as error: raise RuntimeError("provider driver observation is invalid") from error if driver < 12_000: raise RuntimeError("provider driver observation is too old") @@ -316,9 +331,39 @@ def _build_probe(provider_root: Path) -> Path: return probe.resolve() -def generate_candidate(work: Path, provider_root: Path, sm: str) -> tuple[Path, dict[str, Any], Path]: +def _load_candidate_build_module(tensorflow_root: Path) -> Any: + """Load the attested candidate harness without consulting ambient ``ci`` modules.""" + root = tensorflow_root.resolve() + candidate = (root / "ci" / "build_cuda_candidate.py").resolve() + if not candidate.is_relative_to(root) or not candidate.is_file(): + raise RuntimeError("attested TensorFlow candidate build module is unavailable") + spec = importlib.util.spec_from_file_location( + "_rextio_tensorflow_cuda_e3_candidate_build", candidate + ) + if spec is None or spec.loader is None: + raise RuntimeError("attested TensorFlow candidate build module is not loadable") + module = importlib.util.module_from_spec(spec) + sys.modules[spec.name] = module + try: + spec.loader.exec_module(module) + except Exception: + sys.modules.pop(spec.name, None) + raise + return module + + +def _load_verifier_module() -> Any: + """Load the local offline verifier for package and direct-script invocation.""" + try: + return importlib.import_module("scripts.verify_cuda_e3_evidence") + except ModuleNotFoundError: + return importlib.import_module("verify_cuda_e3_evidence") + + +def generate_candidate( + work: Path, tensorflow_root: Path, provider_root: Path, sm: str +) -> tuple[Path, dict[str, Any], Path]: """Generate once through actual probe-backed provider orchestration.""" - from ci import build_cuda_candidate as build from rextio.analyzer.project_scanner import analyze_project from rextio.build.orchestrator import generate_source_artifact from rextio.config.schema import PluginConfig, RextioConfig @@ -330,6 +375,7 @@ def generate_candidate(work: Path, provider_root: Path, sm: str) -> tuple[Path, from rextio_device_cuda.provider import CudaDeviceProvider from rextio_tensorflow.plugin import PLUGIN_ID + build = _load_candidate_build_module(tensorflow_root) probe = _build_probe(provider_root) probe_before = _sha256(probe) build._write_fixture(work) @@ -356,11 +402,27 @@ def generate_candidate(work: Path, provider_root: Path, sm: str) -> tuple[Path, return rust_dir, validate_and_bind_provider_plan(plan, sm, probe_before), probe -def _load_native_inference(python_dir: Path): +def _load_native_inference(python_dir: Path, extension: Path): + """Import only this run's generated wrapper and exact native extension.""" + python_dir = python_dir.resolve() + extension = extension.resolve() sys.path.insert(0, str(python_dir)) - sys.modules.pop("_rextio_native", None) + for name in tuple(sys.modules): + if name == "_rextio_native" or name == "cuda_app" or name.startswith("cuda_app."): + sys.modules.pop(name, None) + importlib.invalidate_caches() os.environ["REXTIO_NATIVE_MODE"] = "native" from cuda_app.kernels import inference + try: + wrapper_path = Path(inspect.getfile(inference)).resolve() + except (OSError, TypeError) as error: + raise RuntimeError("generated inference wrapper has no inspectable source path") from error + if not wrapper_path.is_relative_to(python_dir): + raise RuntimeError("generated inference wrapper was imported from a different path") + native_module = sys.modules.get("_rextio_native") + native_file = getattr(native_module, "__file__", None) + if not isinstance(native_file, str) or Path(native_file).resolve() != extension: + raise RuntimeError("generated native extension was imported from a different path") return inference @@ -379,14 +441,14 @@ def _assert_tensorflow_runtime_is_loaded(tf: Any) -> None: raise RuntimeError("dynamic loader does not expose dladdr") -def execute_e3(tf: Any, python_dir: Path) -> ExecutionResult: +def execute_e3(tf: Any, python_dir: Path, extension: Path) -> ExecutionResult: """Exercise parity, lifetime, and only the five closed negative boundaries.""" tf.config.set_soft_device_placement(False) tf.config.experimental.set_synchronous_execution(True) if not tf.executing_eagerly() or len(tf.config.list_logical_devices("GPU")) != 1: raise RuntimeError("requires one eager TensorFlow GPU:0") import_generated_inference = _load_native_inference - inference = import_generated_inference(python_dir) + inference = import_generated_inference(python_dir, extension) with tf.device("/GPU:0"): x = tf.constant([[1., 2., 3.], [4., 5., 6.], [7., 8., 9.], [2., 1., 0.]], tf.float32) weight = tf.constant([[1., 0.], [0., 1.], [1., 1.]], tf.float32) @@ -456,10 +518,7 @@ def build_payload(*, source: dict[str, Any], environment: dict[str, Any], toolch def main() -> int: """Collect one new self-attested evidence envelope in the closed schema.""" - try: - from scripts import verify_cuda_e3_evidence as verifier - except ModuleNotFoundError: - import verify_cuda_e3_evidence as verifier + verifier = _load_verifier_module() args = build_parser().parse_args() validate_requested_paths_and_values(args, verifier.SMS) toolchain = validate_host() @@ -479,11 +538,13 @@ def main() -> int: _assert_tensorflow_runtime_is_loaded(tf) args.work_dir.mkdir(parents=False) try: - rust_dir, bindings, probe = generate_candidate(args.work_dir, provider_root, args.sm) + rust_dir, bindings, probe = generate_candidate( + args.work_dir, tensorflow_root, provider_root, args.sm + ) environment = dict(os.environ, PYO3_PYTHON=sys.executable) extension = build_generated_extension(rust_dir, rust_dir.parent / "python", environment) extension_before = _sha256(extension) - result = execute_e3(tf, rust_dir.parent / "python") + result = execute_e3(tf, rust_dir.parent / "python", extension) if _sha256(extension) != extension_before: raise RuntimeError("native extension changed during execution") wheel_root = Path(tf.__file__).resolve().parent.parent diff --git a/tests/test_cuda_e3_manual_harness.py b/tests/test_cuda_e3_manual_harness.py index 045d653..e643a18 100644 --- a/tests/test_cuda_e3_manual_harness.py +++ b/tests/test_cuda_e3_manual_harness.py @@ -129,6 +129,8 @@ def test_cargo_build_is_locked_pinned_and_installs_exact_extension( def fake_run(command, **kwargs): calls.append((command, kwargs["env"])) + if command[2] == "generate-lockfile": + (rust_dir / "Cargo.lock").write_text("# generated\n", encoding="utf-8") return subprocess.CompletedProcess(command, 0, "", "") monkeypatch.setattr(module.subprocess, "run", fake_run) @@ -139,13 +141,101 @@ def fake_run(command, **kwargs): "PYO3_PYTHON": "/venv/bin/python", } installed = module.build_generated_extension(rust_dir, python_dir, environment) - assert calls[0][0][:4] == ["cargo", "+1.93.1", "build", "--locked"] - assert "--release" in calls[0][0] + assert calls[0][0][:3] == ["cargo", "+1.93.1", "generate-lockfile"] assert calls[0][1] == environment + assert calls[1][0][:4] == ["cargo", "+1.93.1", "build", "--locked"] + assert "--release" in calls[1][0] + assert calls[1][1] == environment assert installed == python_dir / "_rextio_native.cpython-311-x86_64-linux-gnu.so" assert installed.read_bytes() == b"native" +def test_cargo_lockfile_must_be_created_before_locked_build( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + module = _module() + rust_dir = tmp_path / "rust" + (rust_dir / "target" / "release").mkdir(parents=True) + (rust_dir / "Cargo.toml").write_text("[package]\nname='x'\n", encoding="utf-8") + calls: list[list[str]] = [] + + def fake_run(command, **kwargs): + calls.append(command) + return subprocess.CompletedProcess(command, 0, "", "") + + monkeypatch.setattr(module.subprocess, "run", fake_run) + with pytest.raises(RuntimeError, match="Cargo.lock"): + module.build_generated_extension(rust_dir, tmp_path / "python", {"PATH": "/usr/bin"}) + assert calls == [["cargo", "+1.93.1", "generate-lockfile", "--manifest-path", str(rust_dir / "Cargo.toml")]] + + +def test_native_loader_evicts_stale_cuda_app_and_binds_current_artifacts( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + module = _module() + python_dir = tmp_path / "python" + package = python_dir / "cuda_app" + package.mkdir(parents=True) + (package / "__init__.py").write_text("", encoding="utf-8") + (package / "kernels.py").write_text( + "import _rextio_native\n\ndef inference():\n return _rextio_native.IDENTITY\n", + encoding="utf-8", + ) + extension = python_dir / "_rextio_native.py" + extension.write_text("IDENTITY = 'current'\n", encoding="utf-8") + + stale_package = type(sys)("cuda_app") + stale_kernels = type(sys)("cuda_app.kernels") + stale_kernels.inference = lambda: "stale" + monkeypatch.setitem(sys.modules, "cuda_app", stale_package) + monkeypatch.setitem(sys.modules, "cuda_app.kernels", stale_kernels) + monkeypatch.setitem(sys.modules, "cuda_app.stale", type(sys)("cuda_app.stale")) + + inference = module._load_native_inference(python_dir, extension) + + assert inference() == "current" + assert Path(inference.__code__.co_filename).resolve().is_relative_to(python_dir) + assert Path(sys.modules["_rextio_native"].__file__).resolve() == extension.resolve() + assert "cuda_app.stale" not in sys.modules + + +def test_native_loader_rejects_native_module_from_another_path( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + module = _module() + python_dir = tmp_path / "python" + package = python_dir / "cuda_app" + package.mkdir(parents=True) + (package / "__init__.py").write_text("", encoding="utf-8") + (package / "kernels.py").write_text("import _rextio_native\ndef inference(): pass\n", encoding="utf-8") + (python_dir / "_rextio_native.py").write_text("", encoding="utf-8") + expected_extension = tmp_path / "expected.py" + expected_extension.write_text("", encoding="utf-8") + + with pytest.raises(RuntimeError, match="different path"): + module._load_native_inference(python_dir, expected_extension) + + +def test_candidate_build_module_is_loaded_from_attested_tensorflow_root( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + module = _module() + tensorflow_root = tmp_path / "tensorflow" + candidate = tensorflow_root / "ci" / "build_cuda_candidate.py" + candidate.parent.mkdir(parents=True) + candidate.write_text("MARKER = 'attested'\n", encoding="utf-8") + shadow = type(sys)("ci") + shadow.build_cuda_candidate = type(sys)("ci.build_cuda_candidate") + shadow.build_cuda_candidate.MARKER = "ambient" + monkeypatch.setitem(sys.modules, "ci", shadow) + monkeypatch.setitem(sys.modules, "ci.build_cuda_candidate", shadow.build_cuda_candidate) + + loaded = module._load_candidate_build_module(tensorflow_root) + + assert loaded.MARKER == "attested" + assert Path(loaded.__file__).resolve() == candidate.resolve() + + def test_provider_observations_are_validated_and_bound_to_hashes() -> None: module = _module() profile = {"target_triple": module.TARGET} @@ -355,6 +445,7 @@ def test_source_contains_explicit_tensorflow_before_extension_and_provenance_gua '"readelf"', "dladdr", "sysconfig.get_config_var(\"EXT_SUFFIX\")", - 'sys.modules.pop("_rextio_native"', + 'name == "_rextio_native"', + "importlib.invalidate_caches()", ): assert token in source From ff155595456059396cfe1587a35f81a024516557 Mon Sep 17 00:00:00 2001 From: Rextio Date: Fri, 24 Jul 2026 14:52:13 +0900 Subject: [PATCH 14/15] fix(cuda): bind E3 verifier to attested checkout --- scripts/certify_cuda_candidate.py | 34 ++++++++++++++++++++-------- tests/test_cuda_e3_manual_harness.py | 25 ++++++++++++++++++++ 2 files changed, 50 insertions(+), 9 deletions(-) diff --git a/scripts/certify_cuda_candidate.py b/scripts/certify_cuda_candidate.py index e0d58da..d8386df 100644 --- a/scripts/certify_cuda_candidate.py +++ b/scripts/certify_cuda_candidate.py @@ -11,7 +11,6 @@ import argparse import gc import hashlib -import importlib import importlib.util import inspect import json @@ -352,12 +351,29 @@ def _load_candidate_build_module(tensorflow_root: Path) -> Any: return module -def _load_verifier_module() -> Any: - """Load the local offline verifier for package and direct-script invocation.""" +def _load_verifier_module(tensorflow_root: Path) -> Any: + """Load only the verifier owned by the attested TensorFlow checkout.""" + root = tensorflow_root.resolve() + verifier = (root / "scripts" / "verify_cuda_e3_evidence.py").resolve() + if not verifier.is_relative_to(root) or not verifier.is_file(): + raise RuntimeError("attested TensorFlow verifier module is unavailable") + spec = importlib.util.spec_from_file_location( + "_rextio_tensorflow_cuda_e3_evidence_verifier", verifier + ) + if spec is None or spec.loader is None: + raise RuntimeError("attested TensorFlow verifier module is not loadable") + module = importlib.util.module_from_spec(spec) + sys.modules[spec.name] = module try: - return importlib.import_module("scripts.verify_cuda_e3_evidence") - except ModuleNotFoundError: - return importlib.import_module("verify_cuda_e3_evidence") + spec.loader.exec_module(module) + except Exception: + sys.modules.pop(spec.name, None) + raise + module_file = getattr(module, "__file__", None) + if not isinstance(module_file, str) or Path(module_file).resolve() != verifier: + sys.modules.pop(spec.name, None) + raise RuntimeError("attested TensorFlow verifier module identity changed") + return module def generate_candidate( @@ -518,12 +534,12 @@ def build_payload(*, source: dict[str, Any], environment: dict[str, Any], toolch def main() -> int: """Collect one new self-attested evidence envelope in the closed schema.""" - verifier = _load_verifier_module() args = build_parser().parse_args() - validate_requested_paths_and_values(args, verifier.SMS) - toolchain = validate_host() roots = tuple(path.resolve() for path in (args.tensorflow_root, args.core_root, args.provider_root)) tensorflow_root, core_root, provider_root = roots + verifier = _load_verifier_module(tensorflow_root) + validate_requested_paths_and_values(args, verifier.SMS) + toolchain = validate_host() validate_destinations_are_outside_checkouts(args, roots) core = checkout_identity(core_root) provider = checkout_identity(provider_root) diff --git a/tests/test_cuda_e3_manual_harness.py b/tests/test_cuda_e3_manual_harness.py index e643a18..c48ec1c 100644 --- a/tests/test_cuda_e3_manual_harness.py +++ b/tests/test_cuda_e3_manual_harness.py @@ -236,6 +236,31 @@ def test_candidate_build_module_is_loaded_from_attested_tensorflow_root( assert Path(loaded.__file__).resolve() == candidate.resolve() +def test_verifier_module_is_loaded_from_attested_tensorflow_root( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + module = _module() + tensorflow_root = tmp_path / "tensorflow" + verifier = tensorflow_root / "scripts" / "verify_cuda_e3_evidence.py" + verifier.parent.mkdir(parents=True) + verifier.write_text("MARKER = 'attested'\nSMS = {'sm_80'}\n", encoding="utf-8") + shadow_scripts = type(sys)("scripts") + shadow_scripts.verify_cuda_e3_evidence = type(sys)("scripts.verify_cuda_e3_evidence") + shadow_scripts.verify_cuda_e3_evidence.MARKER = "ambient-package" + shadow_verifier = type(sys)("verify_cuda_e3_evidence") + shadow_verifier.MARKER = "ambient-script" + monkeypatch.setitem(sys.modules, "scripts", shadow_scripts) + monkeypatch.setitem( + sys.modules, "scripts.verify_cuda_e3_evidence", shadow_scripts.verify_cuda_e3_evidence + ) + monkeypatch.setitem(sys.modules, "verify_cuda_e3_evidence", shadow_verifier) + + loaded = module._load_verifier_module(tensorflow_root) + + assert loaded.MARKER == "attested" + assert Path(loaded.__file__).resolve() == verifier.resolve() + + def test_provider_observations_are_validated_and_bound_to_hashes() -> None: module = _module() profile = {"target_triple": module.TARGET} From 9f5594ad150cad0be8a92b9df68dc636d830df7d Mon Sep 17 00:00:00 2001 From: Rextio Date: Fri, 24 Jul 2026 14:59:47 +0900 Subject: [PATCH 15/15] Harden CUDA E3 manual evidence attestation --- scripts/certify_cuda_candidate.py | 50 +++++++++++++++++----- tests/test_cuda_e3_manual_harness.py | 62 ++++++++++++++++++++++++++++ 2 files changed, 101 insertions(+), 11 deletions(-) diff --git a/scripts/certify_cuda_candidate.py b/scripts/certify_cuda_candidate.py index d8386df..2847f31 100644 --- a/scripts/certify_cuda_candidate.py +++ b/scripts/certify_cuda_candidate.py @@ -24,7 +24,7 @@ import tempfile from dataclasses import dataclass from pathlib import Path -from typing import TYPE_CHECKING, Any, Callable +from typing import TYPE_CHECKING, AbstractSet, Any, Callable if TYPE_CHECKING: import tensorflow as tf # noqa: F401 @@ -53,6 +53,7 @@ "tensorflow_framework": "tensorflow/libtensorflow_framework.so.2", "pywrap_tensorflow_common": "tensorflow/python/lib_pywrap_tensorflow_common.so", } +FROZEN_SMS = frozenset({"sm_60", "sm_61", "sm_70", "sm_72", "sm_75", "sm_80", "sm_86", "sm_87", "sm_89", "sm_90"}) ATOL = 1e-5 RTOL = 1e-5 @@ -117,7 +118,7 @@ def _canonical_hash(value: Any) -> str: ).hexdigest() -def validate_requested_paths_and_values(args: argparse.Namespace, allowed_sms: set[str]) -> None: +def validate_requested_paths_and_values(args: argparse.Namespace, allowed_sms: AbstractSet[str]) -> None: """Reject existing destinations and values outside the frozen contract.""" if args.work_dir.exists(): raise RuntimeError("work-dir must not exist") @@ -166,6 +167,19 @@ def validate_checkout(identity: CheckoutIdentity, expected: str, ancestor: str) subprocess.run(["git", "merge-base", "--is-ancestor", ancestor, identity.head], cwd=identity.root, check=True) +def attest_checkouts( + core_root: Path, provider_root: Path, tensorflow_root: Path, expected_tensorflow_commit: str +) -> tuple[CheckoutIdentity, CheckoutIdentity, CheckoutIdentity]: + """Authenticate all source checkouts before executing checkout-owned helpers.""" + core = checkout_identity(core_root) + provider = checkout_identity(provider_root) + plugin = checkout_identity(tensorflow_root) + validate_checkout(core, CORE_COMMIT, CORE_COMMIT) + validate_checkout(provider, PROVIDER_COMMIT, PROVIDER_COMMIT) + validate_checkout(plugin, expected_tensorflow_commit, BASE_CANDIDATE_COMMIT) + return core, provider, plugin + + def validate_host() -> dict[str, str]: """Require the closed Linux, CPython, and Rust toolchain environment.""" if sys.version_info[:2] != (3, 11) or sys.implementation.name != "cpython" or sys.prefix == sys.base_prefix: @@ -274,6 +288,20 @@ def read_build_id(path: Path) -> str | None: return match.group(1).lower() if match else None +def mapped_canonical_paths(maps: str) -> set[Path]: + """Return exact canonical non-deleted file paths recorded in ``/proc/self/maps``.""" + paths: set[Path] = set() + for line in maps.splitlines(): + fields = line.split(maxsplit=5) + if len(fields) != 6: + continue + mapped = fields[5] + if not mapped.startswith("/") or mapped.endswith(" (deleted)"): + continue + paths.add(Path(mapped).resolve()) + return paths + + def capture_runtime_images(wheel_root: Path, maps: str, read_build_id: Callable[[Path], str | None] = read_build_id) -> list[dict[str, Any]]: """Bind hashes to the three wheel DSOs actually mapped by this process.""" canonical = { @@ -281,10 +309,11 @@ def capture_runtime_images(wheel_root: Path, maps: str, read_build_id: Callable[ "tensorflow_framework": wheel_root / "tensorflow" / "libtensorflow_framework.so.2", "pywrap_tensorflow_common": wheel_root / "tensorflow" / "python" / "lib_pywrap_tensorflow_common.so", } + mapped_paths = mapped_canonical_paths(maps) rows: list[dict[str, Any]] = [] for role, path in canonical.items(): resolved = path.resolve() - if not path.is_file() or str(resolved) not in maps: + if not path.is_file() or resolved not in mapped_paths: raise RuntimeError(f"expected TensorFlow runtime image is not mapped: {role}") relative = path.relative_to(wheel_root).as_posix() build_id = read_build_id(path) @@ -537,16 +566,15 @@ def main() -> int: args = build_parser().parse_args() roots = tuple(path.resolve() for path in (args.tensorflow_root, args.core_root, args.provider_root)) tensorflow_root, core_root, provider_root = roots + validate_requested_paths_and_values(args, FROZEN_SMS) + validate_destinations_are_outside_checkouts(args, roots) + core, provider, plugin = attest_checkouts( + core_root, provider_root, tensorflow_root, args.expected_tensorflow_commit + ) verifier = _load_verifier_module(tensorflow_root) - validate_requested_paths_and_values(args, verifier.SMS) + if verifier.SMS != FROZEN_SMS: + raise RuntimeError("attested verifier allowed architectures changed") toolchain = validate_host() - validate_destinations_are_outside_checkouts(args, roots) - core = checkout_identity(core_root) - provider = checkout_identity(provider_root) - plugin = checkout_identity(tensorflow_root) - validate_checkout(core, CORE_COMMIT, CORE_COMMIT) - validate_checkout(provider, PROVIDER_COMMIT, PROVIDER_COMMIT) - validate_checkout(plugin, args.expected_tensorflow_commit, BASE_CANDIDATE_COMMIT) _add_sources(tensorflow_root, core_root, provider_root) import tensorflow as tf if tf.__version__ != "2.21.0": diff --git a/tests/test_cuda_e3_manual_harness.py b/tests/test_cuda_e3_manual_harness.py index c48ec1c..e98f0fe 100644 --- a/tests/test_cuda_e3_manual_harness.py +++ b/tests/test_cuda_e3_manual_harness.py @@ -261,6 +261,47 @@ def test_verifier_module_is_loaded_from_attested_tensorflow_root( assert Path(loaded.__file__).resolve() == verifier.resolve() +def test_main_attests_all_checkouts_before_loading_verifier( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + module = _module() + args = argparse.Namespace( + tensorflow_root=tmp_path / "tensorflow", + core_root=tmp_path / "core", + provider_root=tmp_path / "provider", + expected_tensorflow_commit="a" * 40, + work_dir=tmp_path / "work", + output=tmp_path / "evidence.json", + sm="sm_80", + ) + order: list[str] = [] + + class Parser: + def parse_args(self): + return args + + def attest(*_args): + order.append("attest") + return (object(), object(), object()) + + def load_verifier(_root): + order.append("load-verifier") + return type("Verifier", (), {"SMS": module.FROZEN_SMS})() + + def stop_after_loading(): + order.append("validate-host") + raise RuntimeError("stop test") + + monkeypatch.setattr(module, "build_parser", lambda: Parser()) + monkeypatch.setattr(module, "attest_checkouts", attest) + monkeypatch.setattr(module, "_load_verifier_module", load_verifier) + monkeypatch.setattr(module, "validate_host", stop_after_loading) + + with pytest.raises(RuntimeError, match="stop test"): + module.main() + assert order == ["attest", "load-verifier", "validate-host"] + + def test_provider_observations_are_validated_and_bound_to_hashes() -> None: module = _module() profile = {"target_triple": module.TARGET} @@ -348,6 +389,27 @@ def test_runtime_dso_capture_requires_expected_mapped_wheel_images(tmp_path: Pat module.capture_runtime_images(wheel, maps, read_build_id=lambda _: None) +def test_runtime_dso_capture_rejects_suffix_and_deleted_map_entries(tmp_path: Path) -> None: + module = _module() + wheel = tmp_path.resolve() + paths = ( + wheel / "tensorflow" / "libtensorflow_cc.so.2", + wheel / "tensorflow" / "libtensorflow_framework.so.2", + wheel / "tensorflow" / "python" / "lib_pywrap_tensorflow_common.so", + ) + for path in paths: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_bytes(path.name.encode()) + + suffix_maps = "\n".join(f"7f-8 r-xp 0 00:00 0 {path}.shadow" for path in paths) + with pytest.raises(RuntimeError, match="mapped"): + module.capture_runtime_images(wheel, suffix_maps, read_build_id=lambda _: "deadbeef") + + deleted_maps = "\n".join(f"7f-8 r-xp 0 00:00 0 {path} (deleted)" for path in paths) + with pytest.raises(RuntimeError, match="mapped"): + module.capture_runtime_images(wheel, deleted_maps, read_build_id=lambda _: "deadbeef") + + def test_execution_payload_records_only_observed_boundaries_and_no_profiler_claims() -> None: module = _module() result = module.ExecutionResult(