diff --git a/.github/workflows/performance-guard.yml b/.github/workflows/performance-guard.yml index a207166c0..ab19a87cf 100644 --- a/.github/workflows/performance-guard.yml +++ b/.github/workflows/performance-guard.yml @@ -137,11 +137,14 @@ jobs: retention-days: 30 capture-ab: - # A real debug-heavy link, where the synthetic guard workloads above are too small to show - # write-phase work: Wild itself built with fat LTO and full debug info (one ~250 MB object), - # captured with WILD_SAVE_BASE and relinked by the merge base and the PR head. The PR head's - # output must also match upstream wild's byte for byte. Runs on PRs labelled `performance`. - name: Real full-LTO link A/B (merge base vs PR head, upstream byte identity) + # Real links, where the synthetic guard workloads above are too small to show write-phase + # work. Two of the benchmarks from benchmarks/performance_guard/README.md fit a hosted runner: + # `wild-lto-debug` (Wild built with fat LTO and full debug info, one ~250 MB object) and + # `wild` (Wild's ordinary release build). Both are captured with WILD_SAVE_BASE and relinked + # by the merge base and the PR head, interleaved, into the per-benchmark statistics table that + # upstream performance PRs carry. The PR head's output must also match upstream wild's byte for + # byte. Runs on PRs labelled `performance`. + name: Real link A/B tables (merge base vs PR head, upstream byte identity) if: contains(github.event.pull_request.labels.*.name, 'performance') runs-on: ubuntu-24.04 timeout-minutes: 60 @@ -198,77 +201,89 @@ jobs: cp "$tree/target/release/wild" "$RUNNER_TEMP/bins/$tree-wild" done sha256sum "$RUNNER_TEMP/bins/"* - - name: Capture a fat-LTO debug link of Wild + - name: Test the PR's performance tooling + shell: bash + env: + PYTHONDONTWRITEBYTECODE: "1" + run: python3 -m unittest discover -s candidate/benchmarks/performance_guard -p 'test_*.py' -v + - name: Capture Wild's own links (wild-lto-debug and wild) shell: bash run: | set -euo pipefail - # Plain cargo: soldr's compile cache isn't wanted for a one-off fat-LTO build. - cd candidate - env -u CLICOLOR_FORCE -u FORCE_COLOR \ - RUSTUP_TOOLCHAIN="$RUST_TOOLCHAIN" \ + shopt -s inherit_errexit + # Plain cargo: soldr's compile cache isn't wanted for one-off capture builds. + # Prints the save-dir of the wild binary's own link (build scripts are captured too). + capture_wild() { + local name=$1 + shift + (cd candidate && env -u CLICOLOR_FORCE -u FORCE_COLOR \ + RUSTUP_TOOLCHAIN="$RUST_TOOLCHAIN" \ + CARGO_TARGET_DIR="$RUNNER_TEMP/target-$name" \ + CARGO_TARGET_X86_64_UNKNOWN_LINUX_GNU_LINKER=clang \ + RUSTFLAGS="-Clink-arg=--ld-path=$RUNNER_TEMP/bins/baseline-wild" \ + WILD_SAVE_BASE="$RUNNER_TEMP/capture-$name" \ + "$@" cargo build --release --package wild-linker --bin wild >&2) + local dir + dir=$(dirname "$(grep -l -E '^# Original output file: .*/wild-[0-9a-f]+$' \ + "$RUNNER_TEMP/capture-$name"/*/run-with | head -1)") + du -sh "$dir" >&2 + echo "$dir" + } + lto=$(capture_wild wild-lto-debug env \ CARGO_PROFILE_RELEASE_LTO=fat \ CARGO_PROFILE_RELEASE_DEBUG=full \ - CARGO_PROFILE_RELEASE_CODEGEN_UNITS=1 \ - CARGO_TARGET_DIR="$RUNNER_TEMP/lto-target" \ - CARGO_TARGET_X86_64_UNKNOWN_LINUX_GNU_LINKER=clang \ - RUSTFLAGS="-Clink-arg=--ld-path=$RUNNER_TEMP/bins/baseline-wild" \ - WILD_SAVE_BASE="$RUNNER_TEMP/capture" \ - cargo build --release --package wild-linker --bin wild - # The biggest capture is the wild binary's own link. - capture=$(du -s "$RUNNER_TEMP/capture"/*/ | sort -n | tail -1 | cut -f2) - echo "CAPTURE=$capture" >> "$GITHUB_ENV" - du -sh "$capture" - - name: Prepare a job-private perf collector - shell: bash - run: | - set -uo pipefail - # Follows zackees/mimalloc-pprof ci/perf_access.py: never relax perf_event_paranoid; - # grant CAP_PERFMON to a private copy of the kernel-matched perf ELF instead. - sudo apt-get install -y -qq "linux-tools-$(uname -r)" linux-tools-common >/dev/null 2>&1 || true - elf="/usr/lib/linux-tools/$(uname -r)/perf" - perf_bin=perf - if [ -x "$elf" ]; then - install -d -m 0700 "$RUNNER_TEMP/perf" - cp "$elf" "$RUNNER_TEMP/perf/perf-private" - sudo -n chown "root:$(id -g)" "$RUNNER_TEMP/perf/perf-private" && - sudo -n chmod 0550 "$RUNNER_TEMP/perf/perf-private" && - sudo -n setcap cap_perfmon=ep "$RUNNER_TEMP/perf/perf-private" && - perf_bin="$RUNNER_TEMP/perf/perf-private" - fi - echo "PERF_BIN=$perf_bin" >> "$GITHUB_ENV" - echo "perf_event_paranoid=$(cat /proc/sys/kernel/perf_event_paranoid) perf=$perf_bin" - - name: Paired A/B with upstream byte identity + CARGO_PROFILE_RELEASE_CODEGEN_UNITS=1) + plain=$(capture_wild wild env) + { + echo "CAPTURE_WILD_LTO_DEBUG=$lto" + echo "CAPTURE_WILD=$plain" + } >> "$GITHUB_ENV" + - name: Paired A/B tables with upstream byte identity shell: bash run: | set -euo pipefail mkdir -p evidence + base=${{ steps.revisions.outputs.base }} + cand=${{ steps.revisions.outputs.candidate }} { - echo "## Real full-LTO link A/B" + echo "## Real link A/B" echo - echo "- Baseline (merge base): \`${{ steps.revisions.outputs.base }}\`" - echo "- Candidate (PR head): \`${{ steps.revisions.outputs.candidate }}\`" - echo "- Upstream reference: \`${{ steps.revisions.outputs.upstream }}\`" + echo "- #1 baseline (merge base): \`$base\`" + echo "- #2 candidate (PR head): \`$cand\`" + echo "- Upstream reference for byte identity: \`${{ steps.revisions.outputs.upstream }}\`" echo "- Runner: $(nproc) CPUs, $(grep -m1 'model name' /proc/cpuinfo | cut -d: -f2)" echo + echo "Tables compare against the fork's merge base, not upstream main; see" + echo "benchmarks/performance_guard/README.md for the upstream-PR table." + echo } > evidence/summary.md - for mode in none fast; do + status=0 + for bench in wild-lto-debug wild; do + case $bench in + wild-lto-debug) capture=$CAPTURE_WILD_LTO_DEBUG ;; + wild) capture=$CAPTURE_WILD ;; + esac + printf '### %s\n\n' "$bench" >> evidence/summary.md + ab=(python3 candidate/benchmarks/performance_guard/capture_ab.py + --capture "$capture" --benchmark "$bench" + --baseline "$RUNNER_TEMP/bins/baseline-wild" --baseline-label "merge-base ${base:0:12}" + --candidate "$RUNNER_TEMP/bins/candidate-wild" --candidate-label "pr-head ${cand:0:12}" + --reference "$RUNNER_TEMP/bins/upstream-wild") + # The table: the capture's own arguments, with --no-fork so rusage is the linker's. + "${ab[@]}" --extra=--no-fork --pairs 30 --max-pairs 5000 --budget-seconds 240 \ + --output "evidence/table-$bench.json" --markdown evidence/summary.md || status=1 # Ubuntu's clang passes --build-id, so the capture has one; a later flag overrides it. - extra="--build-id=$mode" - reference=(--reference "$RUNNER_TEMP/bins/upstream-wild") - printf '### --build-id=%s\n\n' "$mode" >> evidence/summary.md - python3 candidate/benchmarks/performance_guard/capture_ab.py \ - --capture "$CAPTURE" \ - --baseline "$RUNNER_TEMP/bins/baseline-wild" \ - --candidate "$RUNNER_TEMP/bins/candidate-wild" \ - "${reference[@]}" \ - --extra="$extra" \ - --perf "$PERF_BIN" \ - --pairs 30 \ - --output "evidence/capture-ab-$mode.json" \ - --markdown evidence/summary.md + for mode in none fast; do + printf '\nIdentity with `--build-id=%s`:\n\n' "$mode" >> evidence/summary.md + "${ab[@]}" --extra="--build-id=$mode --no-fork" --pairs 0 --warmups 1 \ + --output "evidence/identity-$bench-$mode.json" --markdown evidence/summary.md || status=1 + done echo >> evidence/summary.md done cat evidence/summary.md >> "$GITHUB_STEP_SUMMARY" + python3 candidate/benchmarks/performance_guard/pr_table.py evidence/table-*.json \ + > evidence/pr-table.md || status=1 + exit $status - name: Upload A/B evidence if: always() uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0 diff --git a/AGENTS.md b/AGENTS.md index 9de190c3d..1a16fe40e 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -15,3 +15,8 @@ - `main` is the fork: its own commits rebased on `upstream`. To take new upstream work, fast-forward `upstream`, then rebase `main` onto it. - PR branches come off `main` and target `main`. - The performance guard requires a PR's output to match wild built from the newest `upstream` commit the PR contains, so keep `upstream` exactly equal to what `main` is rebased on. + +## Performance PRs + +- Every performance PR proposed to `wild-linker/wild` must include the per-benchmark statistics table described in `benchmarks/performance_guard/README.md` ("Upstream performance PRs"), comparing the exact PR head against the exact upstream `main` it is based on. +- Generate the tables with `benchmarks/performance_guard/capture_ab.py` and paste the output of `pr_table.py`; never hand-type or edit the numbers. Fork CI's tables (the `performance` label) compare against the fork's merge base and are not a substitute. diff --git a/BENCHMARKING.md b/BENCHMARKING.md index 24c47c39d..ef5842377 100644 --- a/BENCHMARKING.md +++ b/BENCHMARKING.md @@ -389,3 +389,9 @@ cargo run --bin benchmark-runner -- \ bench --config benchmarks/ryzen-9955hx.toml --saves ~/save --tmp /ram/%fs/out linker1 linker2 linker3 cargo run --bin benchmark-runner -- report ``` + +### Tables for performance pull requests + +This fork's requirements for performance PRs sent upstream (a per-benchmark statistics table +comparing the PR head against upstream `main`, and the tool that produces it) are in +[benchmarks/performance_guard/README.md](benchmarks/performance_guard/README.md#upstream-performance-prs-the-per-benchmark-table). diff --git a/benchmarks/performance_guard/README.md b/benchmarks/performance_guard/README.md index 47f20aac6..5544c2805 100644 --- a/benchmarks/performance_guard/README.md +++ b/benchmarks/performance_guard/README.md @@ -75,3 +75,122 @@ python3 benchmarks/performance_guard/compare_revisions.py \ ``` Results are written to `results.json` and `results.md` in `--output`. + +## Upstream performance PRs: the per-benchmark table + +Every performance PR we propose to `wild-linker/wild` must include, for each benchmark it +touches, a table in the format the maintainer uses to review performance changes +([wild#2621](https://github.com/wild-linker/wild/pull/2621#issuecomment-5910436060); these are +its `wild-lto-debug` numbers, with this tool's header lines): + +``` +#1 baseline upstream-main +#2 candidate pr-head + metric mean CI ± SD n delta vs #1 (95% CI) + wall 81.263 ms 0.09% 1.87% 1771 -6.60% [-6.71%, -6.49%] + user 687.344 ms 0.18% 3.83% 1771 -16.37% [-16.56%, -16.18%] + system 173.787 ms 0.50% 10.79% 1771 -2.31% [-3.01%, -1.60%] + peak RSS 467.30 MiB 0.01% 0.16% 1771 +0.07% [+0.06%, +0.08%] + cycles 1.313 G~ 0.29% 6.31% 1771 -17.14% [-17.42%, -16.87%] + instructions 1.370 G~ 0.22% 4.76% 1771 -34.47% [-34.67%, -34.27%] + cache-references 94.601 M~ 3.42% 73.34% 1771 -23.56% [-27.09%, -20.02%] + cache-misses 18.960 M~ 2.95% 63.36% 1771 -16.94% [-20.43%, -13.44%] + branches 209.206 M~ 0.46% 9.92% 1771 -34.43% [-34.82%, -34.05%] + branch-misses 63.827 M~ 4.76% 102.23% 1771 -16.87% [-22.05%, -11.70%] +``` + +Rules: + +- `#1` is the exact current upstream `main` the PR branch is based on, and `#2` is the exact PR + head. Rebuild both after every rebase or push; old numbers are context only. Both are built the + same way (`--release`, same toolchain, no debug info). +- The table comes from `capture_ab.py` / `pr_table.py`. Paste their output; never hand-type or + edit numbers. +- Include every benchmark below that the change can plausibly affect, and at least `wild` and + `wild-lto-debug`. For a benchmark you couldn't run, say so rather than leaving it out silently. +- Outputs must be identical apart from the `.comment` provenance record and build ID; the tool + checks this for every table. + +The maintainer's own tool that prints this layout isn't published. It isn't in upstream's +`benchmarks/runner` (whose `report` uses 99% intervals and no hardware counters), and his earlier +reviews used [poop](https://github.com/andrewrk/poop). `capture_ab.py` reproduces the layout: + +- Runs are warmed up, then interleaved in ABBA pairs until `--pairs` is reached and, after that, + until `--budget-seconds` of measuring is spent (capped by `--max-pairs`). Short links need + thousands of runs for CIs of about 0.1%; the maintainer used n = 10,881 for `wild`. +- wall is timed around the fork and exec of `run-with`. user, system and peak RSS come from + `wait4(2)` rusage, so pass `--no-fork` (the default in the commands below) to make the linker do + its own teardown and have peak RSS be the linker's. +- cycles, instructions, cache-references, cache-misses, branches and branch-misses come from + `perf_event_open(2)` counters attached to the child before exec (no `perf` wrapper overhead), + counting user and kernel work where `perf_event_paranoid` allows it and user-only otherwise. + Rows print `n/a` with the reason when the host has no usable PMU, as on GitHub-hosted runners. +- `mean`, `n`, `CI ±` and `SD` describe `#2`. `CI ±` is the 95% Student-t half-width of the mean + and `SD` the sample standard deviation, both as a percentage of the mean. `delta vs #1` is + `mean(#2) / mean(#1) - 1`, with a 95% interval from the delta method on that ratio (standard + error `r * sqrt((se2/m2)^2 + (se1/m1)^2)`, Welch-Satterthwaite degrees of freedom). + +### Running it + +```sh +# Build both binaries the same way. Upstream has no rust-toolchain.toml, so pin the toolchain. +git worktree add ../wild-base upstream/main +git worktree add ../wild-pr +for tree in ../wild-base ../wild-pr; do + (cd "$tree" && RUSTUP_TOOLCHAIN=1.98.1 SOLDR_ALLOW_UNPINNED=1 \ + soldr cargo build --release -p wild-linker --bin wild) +done + +# One benchmark: a save-dir captured with WILD_SAVE_BASE (see below). +python3 benchmarks/performance_guard/capture_ab.py \ + --capture ~/save/wild-lto-debug --benchmark wild-lto-debug \ + --baseline ../wild-base/target/release/wild \ + --baseline-label "upstream-main $(git -C ../wild-base rev-parse --short=12 HEAD)" \ + --candidate ../wild-pr/target/release/wild \ + --candidate-label "pr-head $(git -C ../wild-pr rev-parse --short=12 HEAD)" \ + --extra=--no-fork --pairs 100 --max-pairs 20000 --budget-seconds 600 \ + --output /tmp/pr-tables/wild-lto-debug.json + +# The PR-description markdown for everything measured. +python3 benchmarks/performance_guard/pr_table.py /tmp/pr-tables/*.json +``` + +Measure on a quiet machine (check `/proc/loadavg`), and add `--cpus` to pin both arms to the same +cores if other work can't be stopped. The JSON keeps every raw sample; attach it or keep it with +the PR. + +### Benchmarks + +Captures are made as described in [BENCHMARKING.md](../../BENCHMARKING.md): build the project with +Wild as the linker (`RUSTFLAGS="-Clinker=clang -Clink-arg=--ld-path=$(which wild)"`, or +`-DCMAKE_{EXE,SHARED}_LINKER_FLAGS=--ld-path=$(which wild)` for CMake) and +`WILD_SAVE_BASE=~/save/`, then keep the numbered directory whose `run-with` ends with the +main binary's `# Original output file:`. + +| Benchmark | What is linked | How to capture | Peak RSS / link time (maintainer's machine) | Where | +| --- | --- | --- | --- | --- | +| `wild-lto-debug` | Wild, fat LTO + full debug info | `CARGO_PROFILE_RELEASE_LTO=fat CARGO_PROFILE_RELEASE_DEBUG=full CARGO_PROFILE_RELEASE_CODEGEN_UNITS=1 cargo build --release -p wild-linker --bin wild` | 0.5 GiB / 80 ms | CI and local | +| `wild` | Wild, ordinary release build | `cargo build --release -p wild-linker --bin wild` | 90 MiB / 30 ms | CI and local | +| `rust-analyzer` | rust-analyzer release | `cargo build --release -p rust-analyzer` in rust-lang/rust-analyzer | 0.65 GiB / 120 ms | local | +| `zed-release` | Zed editor release | `cargo build --release -p zed` in zed-industries/zed | 6.5 GiB / 0.6 s | local, big machine | +| `clang-release` | clang from LLVM, `CMAKE_BUILD_TYPE=Release` | CMake + Ninja build of `clang` with the flags above | 0.8 GiB / 100 ms | local, long build | +| `clang-debug` | clang from LLVM, `CMAKE_BUILD_TYPE=Debug` | as above with `Debug` | **25 GiB** / 2.3 s | local, needs 32+ GiB RAM | +| `chrome-android` | Chromium for Android (`target_os="android"`) | `gn gen` + `autoninja` of the main Android library | 4.2 GiB / 1.3 s | local, very long build and large disk | + +Hosted CI (4 CPUs, ~16 GB RAM, 60-minute job) runs only `wild-lto-debug` and `wild`; the others +either can't be built within the job budget or don't fit in memory, so they are measured locally +and pasted from the tool's output. + +### What CI produces + +Labelling a fork PR `performance` runs the `capture-ab` job of +`.github/workflows/performance-guard.yml` (the label only takes effect on the next push or +reopen). It builds the merge base, the PR head and the newest upstream commit the PR contains, +captures `wild-lto-debug` and `wild`, and for each prints this table (merge base as `#1`, PR head +as `#2`, about four minutes of interleaved pairs each) to the job summary. The JSON reports and +`pr-table.md` are in the `capture-ab-` artifact. It also checks that the PR head's output is +byte-identical to upstream wild's with `--build-id=none` and `--build-id=fast`. + +These CI tables compare against the fork's merge base, so they show what the fork PR changes. For +the upstream PR, rerun the tool with upstream `main` as `#1` and the upstream-bound branch as +`#2`, as above. Hosted runners have no hardware counters, so the counter rows there read `n/a`. diff --git a/benchmarks/performance_guard/capture_ab.py b/benchmarks/performance_guard/capture_ab.py index 9e924e44e..eb0e6cdc8 100644 --- a/benchmarks/performance_guard/capture_ab.py +++ b/benchmarks/performance_guard/capture_ab.py @@ -1,79 +1,143 @@ #!/usr/bin/env python3 """Paired A/B of two Wild binaries relinking one `WILD_SAVE_DIR` capture. -Runs are interleaved (ABBA order) after warmups. Each run is wrapped in `perf stat` when it is -available, so every sample carries hardware counters alongside wall time. Outputs of the first -pair are compared after normalising the Wild provenance record in `.comment`, so a speedup can't -come from linking something different. +Runs are interleaved (ABBA order) after warmups. Every sample records wall time, user and system +time and peak RSS (wait4(2) rusage; pass `--extra=--no-fork` so that is the linker's own), plus the +hardware counters perf_event_open(2) allows. Outputs of the warmup runs are compared after +normalising the Wild provenance record in `.comment` and the build ID, so a speedup can't come +from linking something different. + +The report embeds the per-benchmark statistics table from `pr_table.py`, the format upstream +performance PRs must carry; see README.md. """ from __future__ import annotations import argparse +import ctypes import json import os import platform -import shutil +import signal import statistics +import struct import subprocess import sys import tempfile import time from pathlib import Path -EVENTS = ( - "task-clock", - "page-faults", - "context-switches", - "cycles:u", - "instructions:u", - "branches:u", - "branch-misses:u", - "L1-dcache-load-misses:u", - "dTLB-load-misses:u", +import pr_table + +PERF_TYPE_HARDWARE = 0 +# Hardware counters in table order, with their PERF_COUNT_HW_* config values. +COUNTERS = ( + ("cycles", 0), + ("instructions", 1), + ("cache-references", 2), + ("cache-misses", 3), + ("branches", 4), + ("branch-misses", 5), ) +_SYSCALL_PERF_EVENT_OPEN = {"x86_64": 298, "aarch64": 241, "riscv64": 241} +_DISABLED, _INHERIT, _EXCLUDE_KERNEL, _EXCLUDE_HV, _ENABLE_ON_EXEC = 1, 2, 1 << 5, 1 << 6, 1 << 12 +_READ_TIMES = 1 | 2 # PERF_FORMAT_TOTAL_TIME_ENABLED | PERF_FORMAT_TOTAL_TIME_RUNNING + + +class _PerfEventAttr(ctypes.Structure): + # PERF_ATTR_SIZE_VER0, which every kernel accepts. + _fields_ = [ + ("type", ctypes.c_uint32), + ("size", ctypes.c_uint32), + ("config", ctypes.c_uint64), + ("sample_period", ctypes.c_uint64), + ("sample_type", ctypes.c_uint64), + ("read_format", ctypes.c_uint64), + ("flags", ctypes.c_uint64), + ("wakeup_events", ctypes.c_uint32), + ("bp_type", ctypes.c_uint32), + ("config1", ctypes.c_uint64), + ] + + +class Counters: + """Hardware counters for one linker run, opened with perf_event_open(2). + + Counters are attached to the forked child before it execs and enabled on exec, so they count + the linker (and anything it forks) but not this script or a `perf` wrapper. Unavailable + counters (no PMU on a VM, perf_event_paranoid too strict) leave `available` empty and + `reason` saying why; those rows are then printed as n/a. + """ + + def __init__(self, enabled: bool = True): + self.available: list[tuple[str, int]] = [] + self.scope = "" + self.reason = "disabled with --no-counters" + self.max_multiplexing = 1.0 + if not enabled: + return + number = _SYSCALL_PERF_EVENT_OPEN.get(platform.machine()) + if number is None: + self.reason = f"perf_event_open not wired up for {platform.machine()}" + return + self._syscall = ctypes.CDLL(None, use_errno=True).syscall + self._number = number + errors = {} + for scope, flags in (("user+kernel", 0), ("user only", _EXCLUDE_KERNEL | _EXCLUDE_HV)): + self._flags = flags + usable = [] + for name, config in COUNTERS: + try: + os.close(self._open(0, config, probe=True)) + usable.append((name, config)) + except OSError as error: + errors[name] = error.strerror + if usable: + self.available, self.scope = usable, scope + self.reason = "" + break + missing = [name for name, _ in COUNTERS if name not in dict(self.available)] + if missing: + detail = sorted(set(errors[name] for name in missing if name in errors)) + self.reason = "hardware counter unavailable: " + ", ".join(detail or ["unknown"]) + def _open(self, pid: int, config: int, probe: bool = False) -> int: + attr = _PerfEventAttr(type=PERF_TYPE_HARDWARE, size=ctypes.sizeof(_PerfEventAttr), + config=config, read_format=_READ_TIMES) + attr.flags = self._flags | _DISABLED | (0 if probe else _INHERIT | _ENABLE_ON_EXEC) + fd = self._syscall(ctypes.c_long(self._number), ctypes.byref(attr), ctypes.c_int(pid), + ctypes.c_int(-1), ctypes.c_int(-1), ctypes.c_ulong(0)) + if fd < 0: + code = ctypes.get_errno() + raise OSError(code, os.strerror(code)) + return fd -PERF = "perf" - - -def perf_available() -> list[str]: - """Returns the subset of EVENTS this machine can count.""" - if shutil.which(PERF) is None: - return [] - usable = [] - for event in EVENTS: - result = subprocess.run( - [PERF, "stat", "-x", ",", "-e", event, "--", "true"], - capture_output=True, - text=True, - check=False, - ) - line = next((l for l in result.stderr.splitlines() if event in l), "") - if result.returncode == 0 and line and not line.startswith(" dict[str, float]: - counters = {} - for line in path.read_text().splitlines(): - fields = line.split(",") - if len(fields) < 3 or not fields[0] or fields[0].startswith("<"): - continue + def attach(self, pid: int) -> dict[str, int]: + fds = {} try: - value = float(fields[0]) - except ValueError: - continue - # task-clock is msec in older perf and nanoseconds (under various unit names) in newer. - if fields[2] == "task-clock" and fields[1] != "msec": - value /= 1e6 - counters[fields[2]] = value - return counters - - -def run_once(capture: Path, wild: Path, out: Path, events: list[str], extra: list[str], - cpus: str | None, workdir: Path) -> dict[str, float]: + for name, config in self.available: + fds[name] = self._open(pid, config) + except OSError: + for fd in fds.values(): + os.close(fd) + raise + return fds + + def read(self, fds: dict[str, int]) -> dict[str, float]: + values = {} + for name, fd in fds.items(): + value, enabled, running = struct.unpack("QQQ", os.read(fd, 24)) + os.close(fd) + if running: + # Scale up if the kernel had to multiplex more counters than the PMU has. + self.max_multiplexing = max(self.max_multiplexing, enabled / running) + values[name] = value * enabled / running + return values + + +def run_once(capture: Path, wild: Path, out: Path, counters: Counters | None, extra: list[str], + cpus: str | None) -> dict[str, float]: + """One link: wall time, user/system time and peak RSS from wait4(2), plus counters.""" env = dict(os.environ) env.pop("BASH_ENV", None) env["OUT"] = str(out) @@ -82,15 +146,39 @@ def run_once(capture: Path, wild: Path, out: Path, events: list[str], extra: lis command += ["--", *extra] if cpus: command = ["taskset", "-c", cpus, *command] - perf_out = workdir / "perf.csv" - if events: - command = [PERF, "stat", "-x", ",", "-o", str(perf_out), "-e", ",".join(events), "--", - *command] + ready_read, ready_write = os.pipe() + pid = os.fork() + if pid == 0: # pragma: no cover - child + try: + os.close(ready_write) + os.read(ready_read, 1) + devnull = os.open(os.devnull, os.O_WRONLY) + os.dup2(devnull, 1) + os.execvpe(command[0], command, env) + finally: + os._exit(127) + os.close(ready_read) + try: + fds = counters.attach(pid) if counters and counters.available else {} + except BaseException: + os.kill(pid, signal.SIGKILL) + os.waitpid(pid, 0) + raise start = time.perf_counter() - subprocess.run(command, env=env, check=True, stdout=subprocess.DEVNULL) - sample = {"wall_ms": (time.perf_counter() - start) * 1000} - if events: - sample.update(parse_perf(perf_out)) + os.write(ready_write, b"x") + os.close(ready_write) + _, status, usage = os.wait4(pid, 0) + wall_ms = (time.perf_counter() - start) * 1000 + sample = counters.read(fds) if fds else {} + code = os.waitstatus_to_exitcode(status) + if code != 0: + raise subprocess.CalledProcessError(code, command) + sample.update({ + "wall_ms": wall_ms, + "user_ms": usage.ru_utime * 1000, + "sys_ms": usage.ru_stime * 1000, + "max_rss_kib": float(usage.ru_maxrss), + }) return sample @@ -131,29 +219,41 @@ def summarise(samples: list[dict[str, float]]) -> dict[str, dict[str, float]]: return summary +def keep_going(pairs: int, elapsed: float, min_pairs: int, max_pairs: int, + budget_seconds: float) -> bool: + """Whether to run another pair: at least `min_pairs`, then until the time budget is spent.""" + if pairs < min_pairs: + return True + return pairs < max_pairs and elapsed < budget_seconds + + def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--capture", type=Path, required=True, help="WILD_SAVE_DIR directory containing run-with") - parser.add_argument("--baseline", type=Path, required=True) - parser.add_argument("--candidate", type=Path, required=True) + parser.add_argument("--benchmark", help="benchmark name for the table, e.g. wild-lto-debug") + parser.add_argument("--baseline", type=Path, required=True, help="arm #1") + parser.add_argument("--candidate", type=Path, required=True, help="arm #2") + parser.add_argument("--baseline-label", default="baseline", help="e.g. 'upstream-main '") + parser.add_argument("--candidate-label", default="candidate", help="e.g. 'pr-head '") parser.add_argument("--reference", type=Path, help="untimed binary (e.g. upstream wild) whose output the candidate must match") - parser.add_argument("--pairs", type=int, default=30) + parser.add_argument("--pairs", type=int, default=30, + help="minimum measured pairs; 0 only checks output identity") + parser.add_argument("--max-pairs", type=int, help="stop here even within budget (default: --pairs)") + parser.add_argument("--budget-seconds", type=float, default=0, + help="after --pairs, keep adding pairs until this much measuring time is spent") parser.add_argument("--warmups", type=int, default=4) parser.add_argument("--cpus", help="taskset CPU list, e.g. 0-7") parser.add_argument("--extra", default="", help="extra linker arguments, space separated") - parser.add_argument("--no-perf", action="store_true") - parser.add_argument("--perf", default="perf", - help="perf executable, e.g. a private copy granted CAP_PERFMON") + parser.add_argument("--no-counters", action="store_true", help="skip hardware counters") parser.add_argument("--output", type=Path, required=True, help="JSON report path") - parser.add_argument("--markdown", type=Path, help="append a markdown table here") + parser.add_argument("--markdown", type=Path, help="append the result as markdown here") args = parser.parse_args() - global PERF - PERF = args.perf - events = [] if args.no_perf else perf_available() + counters = Counters(enabled=not args.no_counters) extra = args.extra.split() + max_pairs = args.max_pairs if args.max_pairs is not None else args.pairs arms = {"baseline": args.baseline.resolve(), "candidate": args.candidate.resolve()} samples: dict[str, list[dict[str, float]]] = {name: [] for name in arms} @@ -161,64 +261,73 @@ def main() -> int: workdir = Path(tmp) outs = {name: workdir / f"{name}.out" for name in arms} for name, wild in arms.items(): - for _ in range(args.warmups): - run_once(args.capture, wild, outs[name], events, extra, args.cpus, workdir) + for _ in range(max(1, args.warmups)): + run_once(args.capture, wild, outs[name], None, extra, args.cpus) candidate_bytes = normalised(outs["candidate"], workdir) identical = normalised(outs["baseline"], workdir) == candidate_bytes matches_reference = None if args.reference: reference_out = workdir / "reference.out" - run_once(args.capture, args.reference.resolve(), reference_out, [], extra, args.cpus, - workdir) + run_once(args.capture, args.reference.resolve(), reference_out, None, extra, args.cpus) matches_reference = normalised(reference_out, workdir) == candidate_bytes - for pair in range(args.pairs): + start = time.monotonic() + pair = 0 + while keep_going(pair, time.monotonic() - start, args.pairs, max_pairs, + args.budget_seconds): order = list(arms.items()) if pair % 2: order.reverse() for name, wild in order: samples[name].append( - run_once(args.capture, wild, outs[name], events, extra, args.cpus, workdir)) + run_once(args.capture, wild, outs[name], counters, extra, args.cpus)) + pair += 1 - summaries = {name: summarise(s) for name, s in samples.items()} + identity = [f"Baseline and candidate outputs identical apart from `.comment` and build ID: " + f"**{identical}**"] + if matches_reference is not None: + identity.append(f"Candidate output identical to reference apart from `.comment` and build " + f"ID: **{matches_reference}**") report = { - "schema_version": 1, + "schema_version": 2, + "benchmark": args.benchmark, "host": {"platform": platform.platform(), "machine": platform.machine(), "logical_cpus": os.cpu_count()}, "capture": str(args.capture), "extra_args": extra, "cpus": args.cpus, - "pairs": args.pairs, - "events": events, + "labels": {"baseline": args.baseline_label, "candidate": args.candidate_label}, + "pairs": pair, + "counters": {"available": [name for name, _ in counters.available], + "scope": counters.scope, "unavailable_reason": counters.reason, + "max_multiplexing": counters.max_multiplexing}, "outputs_identical": identical, "matches_reference": matches_reference, - "summary": summaries, + "summary": {name: summarise(s) for name, s in samples.items() if s}, "samples": samples, } + lines = [] + if pair: + unavailable = {name: counters.reason for name, _ in COUNTERS} + report["table"] = pr_table.render( + {"label": args.baseline_label, "samples": samples["baseline"]}, + {"label": args.candidate_label, "samples": samples["candidate"]}, + unavailable) + notes = [f"Linker args: `{' '.join(extra) or '(as captured)'}`; {pair} interleaved pairs"] + if counters.available: + notes.append(f"counters: {counters.scope}" + + (f", multiplexed up to {counters.max_multiplexing:.2f}x" + if counters.max_multiplexing > 1.001 else "")) + report["table_notes"] = "; ".join(notes) + "." + lines += [pr_table.markdown(report), ""] + lines += identity args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text(json.dumps(report, indent=1) + "\n") - base, cand = summaries["baseline"], summaries["candidate"] - lines = [ - f"| Metric (median of {args.pairs}) | Baseline | Candidate | Change |", - "| --- | ---: | ---: | ---: |", - ] - for key in ["wall_ms", *[e for e in EVENTS if e in base]]: - if key not in base or key not in cand: - continue - b, c = base[key]["median"], cand[key]["median"] - change = f"{100 * (c - b) / b:+.1f}%" if b else "n/a" - fmt = "{:,.1f}" if key in ("wall_ms", "task-clock") else "{:,.0f}" - lines.append(f"| {key} | {fmt.format(b)} | {fmt.format(c)} | {change} |") - lines.append("") - lines.append(f"Baseline and candidate outputs identical apart from `.comment`: **{identical}**") - if matches_reference is not None: - lines.append(f"Candidate output identical to reference apart from `.comment`: " - f"**{matches_reference}**") - table = "\n".join(lines) - print(table) + text = "\n".join(lines) + print(text) if args.markdown: with args.markdown.open("a") as handle: - handle.write(table + "\n") + handle.write(text + "\n") return 0 if identical and matches_reference is not False else 1 diff --git a/benchmarks/performance_guard/pr_table.py b/benchmarks/performance_guard/pr_table.py new file mode 100644 index 000000000..7ff65ab7d --- /dev/null +++ b/benchmarks/performance_guard/pr_table.py @@ -0,0 +1,169 @@ +#!/usr/bin/env python3 +"""Per-benchmark statistics table for performance PRs. + +Renders the layout wild's maintainer uses when reviewing performance PRs (see +https://github.com/wild-linker/wild/pull/2621#issuecomment-5910436060): + + #1 baseline upstream-main 7a67b54f + #2 candidate pr-head d48eb7d2 + metric mean CI ± SD n delta vs #1 (95% CI) + wall 81.263 ms 0.09% 1.87% 1771 -6.60% [-6.71%, -6.49%] + +`mean`, `CI ±`, `SD` and `n` describe arm #2. `CI ±` is the half-width of the 95% Student-t +interval of the mean and `SD` the sample standard deviation, both as a percentage of the mean. +`delta vs #1` is mean(#2) / mean(#1) - 1 with a 95% interval from the delta method on that ratio +of means (Welch-Satterthwaite degrees of freedom). + +`capture_ab.py` writes this table into its JSON report. Running this module on one or more of +those reports prints the Markdown for a PR description, so tables are never typed by hand: + + python3 benchmarks/performance_guard/pr_table.py evidence/*.json +""" + +from __future__ import annotations + +import argparse +import json +import math +import statistics +import sys +from pathlib import Path + +# (sample key, row label, kind). The order is the order of rows in the table. +METRICS = ( + ("wall_ms", "wall", "ms"), + ("user_ms", "user", "ms"), + ("sys_ms", "system", "ms"), + ("max_rss_kib", "peak RSS", "kib"), + ("cycles", "cycles", "count"), + ("instructions", "instructions", "count"), + ("cache-references", "cache-references", "count"), + ("cache-misses", "cache-misses", "count"), + ("branches", "branches", "count"), + ("branch-misses", "branch-misses", "count"), +) + +HEADER = (" metric mean CI ± SD n" + " delta vs #1 (95% CI)") + +_NORMAL_975 = statistics.NormalDist().inv_cdf(0.975) + + +def t_quantile_975(df: float) -> float: + """Two-sided 95% critical value of Student's t with `df` degrees of freedom.""" + if df <= 0 or math.isnan(df): + return math.nan + exact = {1: 12.706204736174705, 2: 4.302652729749464, 3: 3.182446305284263, + 4: 2.776445105197799} + if df in exact: + return exact[df] + # Cornish-Fisher expansion about the normal quantile; accurate to ~1e-3 from df = 5. + z = _NORMAL_975 + g1 = (z**3 + z) / 4 + g2 = (5 * z**5 + 16 * z**3 + 3 * z) / 96 + g3 = (3 * z**7 + 19 * z**5 + 17 * z**3 - 15 * z) / 384 + g4 = (79 * z**9 + 776 * z**7 + 1482 * z**5 - 1920 * z**3 - 945 * z) / 92160 + return z + g1 / df + g2 / df**2 + g3 / df**3 + g4 / df**4 + + +def describe(values: list[float]) -> dict[str, float]: + """Mean, sample SD, n and the 95% CI half-width of the mean.""" + n = len(values) + mean = statistics.fmean(values) if values else math.nan + sd = statistics.stdev(values) if n > 1 else math.nan + half = t_quantile_975(n - 1) * sd / math.sqrt(n) if n > 1 else math.nan + return {"mean": mean, "sd": sd, "n": n, "ci_half": half} + + +def delta(base: dict[str, float], cand: dict[str, float]) -> tuple[float, float, float]: + """Percentage change of the candidate mean over the baseline mean, with its 95% interval.""" + if not base["mean"] or not cand["mean"] or base["n"] < 2 or cand["n"] < 2: + return math.nan, math.nan, math.nan + ratio = cand["mean"] / base["mean"] + rel_c = (cand["sd"] / math.sqrt(cand["n"]) / cand["mean"]) ** 2 + rel_b = (base["sd"] / math.sqrt(base["n"]) / base["mean"]) ** 2 + se = ratio * math.sqrt(rel_c + rel_b) + if rel_c + rel_b == 0: + df = cand["n"] + base["n"] - 2 + else: + df = (rel_c + rel_b) ** 2 / (rel_c**2 / (cand["n"] - 1) + rel_b**2 / (base["n"] - 1)) + half = t_quantile_975(df) * se + return 100 * (ratio - 1), 100 * (ratio - 1 - half), 100 * (ratio - 1 + half) + + +def format_mean(value: float, kind: str) -> str: + if kind == "ms": + return f"{value:.3f} ms" + if kind == "kib": + return f"{value / 1024:.2f} MiB" + for prefix, scale in (("T", 1e12), ("G", 1e9), ("M", 1e6), ("K", 1e3)): + if abs(value) >= scale: + return f"{value / scale:.3f} {prefix}~" + return f"{value:.3f} ~" + + +def _percent(value: float) -> str: + return "n/a" if math.isnan(value) else f"{value:.2f}%" + + +def format_row(label: str, kind: str, cand: dict[str, float], + change: tuple[float, float, float]) -> str: + """One table row from arm #2's `describe()` statistics and the `delta()` against arm #1.""" + rel = (lambda x: 100 * x / cand["mean"]) if cand["mean"] else (lambda x: math.nan) + pct, lo, hi = change + text = "n/a" if math.isnan(pct) else f"{pct:+.2f}% [{lo:+.2f}%, {hi:+.2f}%]" + return (f" {label:<16}{format_mean(cand['mean'], kind):>17}{_percent(rel(cand['ci_half'])):>12}" + f"{_percent(rel(cand['sd'])):>11}{cand['n']:>8} {text}") + + +def row(label: str, kind: str, base_values: list[float], cand_values: list[float]) -> str: + base, cand = describe(base_values), describe(cand_values) + return format_row(label, kind, cand, delta(base, cand)) + + +def render(baseline: dict, candidate: dict, unavailable: dict[str, str] | None = None) -> str: + """Renders the table for one benchmark. + + `baseline` and `candidate` are `{"label": str, "samples": [ {metric: value} ]}`. Metrics with + no samples in either arm are listed as not measured, with the reason from `unavailable`. + """ + unavailable = unavailable or {} + lines = [f"#1 baseline {baseline['label']}", f"#2 candidate {candidate['label']}", HEADER] + for key, label, kind in METRICS: + b = [s[key] for s in baseline["samples"] if key in s] + c = [s[key] for s in candidate["samples"] if key in s] + if b and c: + lines.append(row(label, kind, b, c)) + else: + reason = unavailable.get(key, "not measured") + lines.append(f" {label:<16}{'n/a':>17} {reason}") + return "\n".join(lines) + + +def markdown(report: dict) -> str: + """One benchmark section for a PR description, from a `capture_ab.py` JSON report.""" + parts = [f"{report.get('benchmark') or report['capture']}:", "", "```", report["table"], "```"] + notes = report.get("table_notes") + if notes: + parts += ["", notes] + return "\n".join(parts) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("reports", nargs="+", type=Path, help="capture_ab.py JSON reports") + args = parser.parse_args() + sections = [] + for path in args.reports: + report = json.loads(path.read_text()) + if "table" not in report: + print(f"{path}: no table (identity-only run?)", file=sys.stderr) + continue + sections.append(markdown(report)) + print("\n\n".join(sections)) + return 0 if sections else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/performance_guard/test_capture_ab.py b/benchmarks/performance_guard/test_capture_ab.py index c5f485f2c..4a9691721 100644 --- a/benchmarks/performance_guard/test_capture_ab.py +++ b/benchmarks/performance_guard/test_capture_ab.py @@ -1,3 +1,5 @@ +import os +import stat import tempfile import unittest from pathlib import Path @@ -6,33 +8,56 @@ class CaptureAbTest(unittest.TestCase): - def test_parse_perf_reads_counters_and_skips_unsupported(self): - with tempfile.TemporaryDirectory() as tmp: - path = Path(tmp) / "perf.csv" - path.write_text( - "# started on Mon Sep 28\n" - "\n" - "455.70,msec,task-clock,455700000,100.00,5.998,CPUs utilized\n" - "1383462596,,cycles:u,455000000,100.00,,\n" - ",,dTLB-load-misses:u,0,100.00,,\n" - ) - counters = capture_ab.parse_perf(path) - self.assertEqual(counters, {"task-clock": 455.70, "cycles:u": 1383462596.0}) - - def test_parse_perf_converts_nanosecond_task_clock_to_ms(self): - with tempfile.TemporaryDirectory() as tmp: - path = Path(tmp) / "perf.csv" - path.write_text("334053834.5,ns,task-clock,334053834,100.00,,\n") - counters = capture_ab.parse_perf(path) - self.assertAlmostEqual(counters["task-clock"], 334.0538345) - def test_summarise_takes_median_per_metric(self): summary = capture_ab.summarise( - [{"wall_ms": 3.0, "cycles:u": 10.0}, {"wall_ms": 1.0}, {"wall_ms": 2.0, "cycles:u": 30.0}] + [{"wall_ms": 3.0, "cycles": 10.0}, {"wall_ms": 1.0}, {"wall_ms": 2.0, "cycles": 30.0}] ) self.assertEqual(summary["wall_ms"]["median"], 2.0) self.assertEqual(summary["wall_ms"]["min"], 1.0) - self.assertEqual(summary["cycles:u"]["median"], 20.0) + self.assertEqual(summary["cycles"]["median"], 20.0) + + def test_keep_going_honours_minimum_budget_and_cap(self): + self.assertTrue(capture_ab.keep_going(3, 100.0, 5, 5, 0)) # below minimum + self.assertFalse(capture_ab.keep_going(5, 0.0, 5, 5, 60)) # cap reached + self.assertTrue(capture_ab.keep_going(5, 10.0, 5, 50, 60)) # within budget + self.assertFalse(capture_ab.keep_going(5, 61.0, 5, 50, 60)) # budget spent + self.assertFalse(capture_ab.keep_going(0, 0.0, 0, 0, 0)) # identity-only + + def test_counters_disabled_report_reason(self): + counters = capture_ab.Counters(enabled=False) + self.assertEqual(counters.available, []) + self.assertIn("--no-counters", counters.reason) + + def test_run_once_measures_a_fake_capture(self): + with tempfile.TemporaryDirectory() as tmp: + capture = Path(tmp) + run_with = capture / "run-with" + # Like a real save-dir script: the linker is the first argument, then `--` and extras. + run_with.write_text('#!/bin/sh\nlinker="$1"; shift; [ "$1" = -- ] && shift\n' + 'exec "$linker" "$OUT" "$@"\n') + linker = capture / "linker" + linker.write_text('#!/bin/sh\necho "$@" > "$1"\n') + for path in (run_with, linker): + path.chmod(path.stat().st_mode | stat.S_IXUSR) + out = capture / "out" + counters = capture_ab.Counters() + sample = capture_ab.run_once(capture, linker, out, counters, ["--no-fork"], None) + self.assertEqual(out.read_text(), f"{out} --no-fork\n") + for key in ("wall_ms", "user_ms", "sys_ms", "max_rss_kib"): + self.assertIn(key, sample) + self.assertGreater(sample["wall_ms"], 0) + self.assertGreater(sample["max_rss_kib"], 0) + for name, _ in counters.available: + self.assertGreater(sample[name], 0, name) + + def test_run_once_raises_on_linker_failure(self): + with tempfile.TemporaryDirectory() as tmp: + capture = Path(tmp) + run_with = capture / "run-with" + run_with.write_text("#!/bin/sh\nexit 3\n") + run_with.chmod(0o755) + with self.assertRaises(capture_ab.subprocess.CalledProcessError): + capture_ab.run_once(capture, Path("/bin/true"), capture / "o", None, [], None) if __name__ == "__main__": diff --git a/benchmarks/performance_guard/test_pr_table.py b/benchmarks/performance_guard/test_pr_table.py new file mode 100644 index 000000000..ab7b00d60 --- /dev/null +++ b/benchmarks/performance_guard/test_pr_table.py @@ -0,0 +1,107 @@ +import json +import math +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +import pr_table + +HERE = Path(__file__).resolve().parent + + +def stats(mean, ci_pct, sd_pct, n): + return {"mean": mean, "ci_half": mean * ci_pct / 100, "sd": mean * sd_pct / 100, "n": n} + + +class FormatTest(unittest.TestCase): + """Rows must match the maintainer's layout character for character (wild#2621).""" + + def test_header_matches(self): + self.assertEqual( + pr_table.HEADER, + " metric mean CI ± SD n delta vs #1 (95% CI)") + + def test_time_row_matches(self): + line = pr_table.format_row("wall", "ms", stats(81.263, 0.09, 1.87, 1771), (-6.60, -6.71, -6.49)) + self.assertEqual( + line, " wall 81.263 ms 0.09% 1.87% 1771 -6.60% [-6.71%, -6.49%]") + + def test_counter_row_matches(self): + line = pr_table.format_row("instructions", "count", stats(89.521e9, 1.51, 10.81, 200), + (-26.73, -28.60, -24.87)) + self.assertEqual( + line, + " instructions 89.521 G~ 1.51% 10.81% 200 -26.73% [-28.60%, -24.87%]") + + def test_rss_row_matches(self): + line = pr_table.format_row("peak RSS", "kib", stats(24874.56 * 1024, 0.0, 0.03, 200), + (0.0, -0.001, 0.01)) + self.assertEqual( + line, " peak RSS 24874.56 MiB 0.00% 0.03% 200 +0.00% [-0.00%, +0.01%]") + + def test_si_prefixes(self): + self.assertEqual(pr_table.format_mean(18.96e6, "count"), "18.960 M~") + self.assertEqual(pr_table.format_mean(1500, "count"), "1.500 K~") + self.assertEqual(pr_table.format_mean(12, "count"), "12.000 ~") + + +class StatisticsTest(unittest.TestCase): + def test_t_quantiles(self): + for df, expected in ((1, 12.706), (2, 4.303), (3, 3.182), (5, 2.571), (10, 2.228), + (30, 2.042), (1000, 1.962)): + self.assertAlmostEqual(pr_table.t_quantile_975(df), expected, delta=2e-3, msg=df) + + def test_describe(self): + d = pr_table.describe([1.0, 2.0, 3.0, 4.0]) + self.assertEqual(d["mean"], 2.5) + self.assertAlmostEqual(d["sd"], 1.2909944, places=6) + self.assertAlmostEqual(d["ci_half"], 3.182446 * 1.2909944 / 2, places=3) + + def test_delta_of_ratio_of_means(self): + base = {"mean": 100.0, "sd": 1.0, "n": 1000} + cand = {"mean": 90.0, "sd": 0.9, "n": 1000} + pct, lo, hi = pr_table.delta(base, cand) + self.assertAlmostEqual(pct, -10.0) + # Each mean is known to 0.1% / sqrt(1000) relative SE, combined * sqrt(2). + half = 1.9623 * 0.9 * math.sqrt(2) * 0.01 / math.sqrt(1000) * 100 + self.assertAlmostEqual(hi - pct, half, places=3) + self.assertAlmostEqual(pct - lo, half, places=3) + + def test_delta_undefined_for_one_sample(self): + self.assertTrue(math.isnan(pr_table.delta({"mean": 1, "sd": math.nan, "n": 1}, + {"mean": 1, "sd": math.nan, "n": 1})[0])) + + +class RenderTest(unittest.TestCase): + def samples(self, scale): + return [{"wall_ms": scale * w, "user_ms": 2.0 * w, "sys_ms": 1.0, "max_rss_kib": 1024.0 * w} + for w in (10.0, 11.0, 12.0, 11.0)] + + def test_unavailable_counters_are_listed(self): + table = pr_table.render({"label": "upstream-main aaaa", "samples": self.samples(1.0)}, + {"label": "pr-head bbbb", "samples": self.samples(0.9)}, + {"cycles": "hardware counter unavailable: No such file or directory"}) + lines = table.splitlines() + self.assertEqual(lines[0], "#1 baseline upstream-main aaaa") + self.assertEqual(lines[1], "#2 candidate pr-head bbbb") + self.assertEqual(len(lines), 3 + len(pr_table.METRICS)) + self.assertTrue(lines[3].startswith(" wall 9.900 ms")) + self.assertIn("-10.00% [", lines[3]) + self.assertEqual(lines[7], " cycles n/a " + "hardware counter unavailable: No such file or directory") + self.assertTrue(lines[8].endswith("n/a not measured")) + + def test_cli_renders_markdown_from_reports(self): + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "r.json" + path.write_text(json.dumps({"benchmark": "wild", "capture": "/c", "table": "T", + "table_notes": "N."})) + out = subprocess.run([sys.executable, str(HERE / "pr_table.py"), str(path)], + capture_output=True, text=True, check=True).stdout + self.assertEqual(out, "wild:\n\n```\nT\n```\n\nN.\n") + + +if __name__ == "__main__": + unittest.main()