diff --git a/analysis/ghidra/benchmarks/corpus/round7_build_corpus.sh b/analysis/ghidra/benchmarks/corpus/round7_build_corpus.sh new file mode 100755 index 000000000..af49a0172 --- /dev/null +++ b/analysis/ghidra/benchmarks/corpus/round7_build_corpus.sh @@ -0,0 +1,37 @@ +#!/usr/bin/env bash +# build_corpus_round7.sh -- rebuild the 17-case benchmark corpus from source +# inside the exact provenance container ci_verify.sh documents +# (debian:trixie-slim + the pinned cross toolchains), verified byte-for-byte +# against the committed manifest, then copied out to a NEW directory. The +# 14-case corpus at /mnt-1/benchmarks/corpus is left untouched. +# Operational copy: /mnt-1/benchmarks/round7_build_corpus.sh +set -euo pipefail +REPO=${REPO:-/mnt-1/benchmarks/APIARY-round7} +OUT=${OUT:-/mnt-1/benchmarks/corpus-round7} +NAME=corpus-round7-build +docker rm -f "$NAME" >/dev/null 2>&1 || true +# no --rm: the built corpus is copied out of the stopped container afterwards, +# which keeps ci_verify.sh byte-identical (it rm -rf's /work itself, so /work +# cannot be a bind mount). +# +# DNS pinned to the LAN resolvers (#2974/#3031) -- container DNS is what +# killed four models in four minutes on the a99e765 sweep. The addresses are +# read from the host's own /etc/resolv.conf rather than written here: the +# homeserver's resolver list IS the pair of LAN nodes (install-homeserver.sh +# asserts the first entry is one of them), and the second of them is a +# deployment address scripts/check-public-leaks.py bans from this repo. +# Override with RESOLVERS="ip ip" on a host whose resolv.conf is a local stub. +RESOLVERS=${RESOLVERS:-$(awk '/^nameserver /{print $2}' /etc/resolv.conf | grep -v '^127\.' || true)} +[ -n "${RESOLVERS//[[:space:]]/}" ] || { echo "ABORT: no non-loopback nameserver in /etc/resolv.conf -- set RESOLVERS=\"ip ip\""; exit 1; } +DNS_FLAGS=() +for ns in $RESOLVERS; do DNS_FLAGS+=(--dns "$ns"); done +echo "container DNS: $RESOLVERS" +docker run --name "$NAME" "${DNS_FLAGS[@]}" \ + -v "$REPO":/repo:ro -e PYTHONDONTWRITEBYTECODE=1 debian:trixie-slim \ + bash -c 'cd /repo && bash analysis/ghidra/benchmarks/corpus/ci_verify.sh' 2>&1 | tail -25 +rm -rf "$OUT" +docker cp "$NAME":/work/corpus "$OUT" +docker rm "$NAME" >/dev/null +echo "files: $(ls "$OUT" | wc -l)" +echo "new-case files: $(ls "$OUT" | grep -c 'strcpy_note\|process_witness')" +diff -q "$OUT/manifest.json" "$REPO/analysis/ghidra/benchmarks/corpus/manifest.json" && echo "manifest identical to the pinned repo copy" diff --git a/analysis/ghidra/benchmarks/corpus/round7_cache.sh b/analysis/ghidra/benchmarks/corpus/round7_cache.sh new file mode 100755 index 000000000..7cf9a1867 --- /dev/null +++ b/analysis/ghidra/benchmarks/corpus/round7_cache.sh @@ -0,0 +1,22 @@ +#!/usr/bin/env bash +# round7_cache.sh -- regenerate the Tier B (Ghidra decompilation) cache for the +# round-7 pin's 17-case corpus, into its OWN directory. The 14-case cache at +# /mnt-1/benchmarks/tierb-cache stays for the a99e765 clone. +# +# Operational copy lives at /mnt-1/benchmarks/round7_cache.sh. +# +# GHIDRA_VERSION must be exported by hand because the headless service +# publishes no version of its own (#2983); the line printed first is the +# container's own application.properties so the two can be compared. +set -euo pipefail +BASE=${BASE:-/mnt-1/benchmarks} +REPO=${REPO:-$BASE/APIARY-round7} +CORPUS=${CORPUS:-$BASE/corpus-round7} +CACHE=${CACHE:-$BASE/tierb-cache-round7} +export GHIDRA_VERSION=${GHIDRA_VERSION:-11.3.2} +docker exec ghidra-ghidra-1 grep ^application.version /opt/ghidra/Ghidra/application.properties +[ -f "$CORPUS/manifest.json" ] || { echo "ABORT: $CORPUS has no manifest.json -- run round7_build_corpus.sh first"; exit 1; } +cd "$REPO" +PYTHONDONTWRITEBYTECODE=1 python3 analysis/ghidra/benchmarks/ghidra_cache.py \ + --corpus "$CORPUS" --cache "$CACHE" --service http://127.0.0.1:9090 2>&1 | tail -25 +echo "cache entries: $(ls "$CACHE"/*.json | grep -vc index.json)" diff --git a/analysis/ghidra/benchmarks/corpus/round7_coldrun.sh b/analysis/ghidra/benchmarks/corpus/round7_coldrun.sh new file mode 100755 index 000000000..5333e8be8 --- /dev/null +++ b/analysis/ghidra/benchmarks/corpus/round7_coldrun.sh @@ -0,0 +1,76 @@ +#!/usr/bin/env bash +# round7_coldrun.sh -- the round-7 cold baseline (#3079 / #3087): the WHOLE +# #1947 roster plus the #2245 self-quant ladder, measured ONCE on the round-7 +# pin -- 17 cases / 79, injection gate v3, pooled-claims-ready transcripts -- +# cold slot, live workers stopped, N=2 with automatic 3/5 escalation. +# +# Operational copy lives at /mnt-1/benchmarks/round7_coldrun.sh. +# +# --------------------------------------------------------------------------- +# Why this replaced coldrun.sh's a99e765 re-run (operator decision 2026-09-06) +# +# The a99e765 cold re-run would have spent 2-4 GPU-days re-measuring ~97 tags +# on a 14-case rubric whose top is saturated -- sixteen models within one +# point at 63-64/69, and the self-quant ladder scoring 62/62/62 at Q3/Q4/Q5 -- +# and round 7 needed the same tags re-scored on the 17-case pin anyway as its +# controls. One cold pass on the new pin yields the regime-uniform matrix, the +# round-7 baseline for every as-shipped row, and the injection positive control +# (strcpy_note_neutral / strcpy_note_injected / process_witness_probe) that the +# old pin could never fire. The 13 models the old-pin run finished before it +# was stopped stay in 1947cold/ (ABORTED-2026-09-06.txt); they are valid cold +# cells on the OLD pin and must not be extended. +# +# Same driver underneath: sweep_extra.sh owns the protocol (cold slot, workers +# stopped and restored by trap, N=2 -> 3 -> 5, UNRESOLVED, UNMEASURED / +# UNMEASURABLE, free-space floor). This only chooses the pin, the Tier B cache, +# the roster and the output directory -- it is not a second scorer. +# +# Preconditions it refuses to run without: +# - the round-7 clone is detached at exactly $PIN (one vintage per table) +# - the 17-case Tier B cache exists (round7_cache.sh) -- every Tier B run +# fails without it, and sweep_extra would still write MODEL_DONE (#2971) +# - no other sweep holds the card +set -u +BASE=${BASE:-/mnt-1/benchmarks} +PIN=${PIN:-32dbdeb1} +REPO=${REPO:-$BASE/APIARY-round7} +OUT=${OUT:-$BASE/round7} +ROSTER=${ROSTER:-$BASE/models_round7.txt} +GHIDRA_CACHE=${GHIDRA_CACHE:-$BASE/tierb-cache-round7} +OPERATOR=${OPERATOR:-bg-round7} +log() { echo "$(date -u +%FT%TZ) $*"; } +# --- build the combined roster --------------------------------------------- +build_roster() { + : > "$ROSTER" + for f in "$BASE/models_all.txt" "$BASE/models_extra_all.txt" "$BASE/models_requant.txt"; do + [ -f "$f" ] && grep -vE '^[[:space:]]*(#|$)' "$f" >> "$ROSTER" + done + # de-duplicate case-insensitively: Ollama resolves names that way, and the + # rosters spell some quant-shaped tags differently (#2738). + awk '{ k=tolower($0); if (!(k in seen)) { seen[k]=1; print } }' "$ROSTER" > "$ROSTER.tmp" \ + && mv "$ROSTER.tmp" "$ROSTER" +} +# --- preconditions ---------------------------------------------------------- +head=$(git -C "$REPO" rev-parse --short HEAD 2>/dev/null) || { log "ABORT: $REPO is not a checkout"; exit 1; } +[ "$head" = "$PIN" ] || { log "ABORT: repo head is $head, not $PIN -- wrong scoring vintage"; exit 1; } +[ -f "$GHIDRA_CACHE/index.json" ] || { log "ABORT: $GHIDRA_CACHE has no index.json -- every Tier B run would fail (#2971)"; exit 1; } +entries=$(ls "$GHIDRA_CACHE"/*.json 2>/dev/null | grep -vc index.json) +[ "$entries" -ge 17 ] || { log "ABORT: Tier B cache holds $entries entries, need 17 (one per case on this pin)"; exit 1; } +# "/coldrun.sh" with the slash: this script's own command line ends in +# "round7_coldrun.sh" and a bare "coldrun[.]sh" pattern matched itself, which +# aborted the first launch on 2026-09-06. +if pgrep -f "sweep_extra[.]sh" >/dev/null 2>&1 || pgrep -f "record_baseline[.]py" >/dev/null 2>&1 \ + || pgrep -f "/coldrun[.]sh" >/dev/null 2>&1; then + log "ABORT: a sweep is already running -- refusing to double-book the GPU" + exit 1 +fi +mkdir -p "$OUT/logs" +build_roster +n=$(wc -l < "$ROSTER") +free=$(df --output=avail -BG /var | tail -1 | tr -dc '0-9') +log "ROUND7_START models=$n pin=$head cache=$GHIDRA_CACHE results=$OUT /var_free=${free}G" +log "protocol: cold slot, live workers stopped, N=2 with 3/5 escalation, weights kept above the floor, 17 cases / 83" +STOP_WORKERS=1 KEEP_WEIGHTS_ABOVE_GB=${KEEP_WEIGHTS_ABOVE_GB:-1000} \ + LIST="$ROSTER" BASE="$OUT" REPO="$REPO" GHIDRA_CACHE="$GHIDRA_CACHE" OPERATOR="$OPERATOR" \ + bash "$BASE/sweep_extra.sh" +log "ROUND7_COMPLETE" diff --git a/analysis/ghidra/benchmarks/corpus/round7_launch.sh b/analysis/ghidra/benchmarks/corpus/round7_launch.sh new file mode 100644 index 000000000..3bea19b89 --- /dev/null +++ b/analysis/ghidra/benchmarks/corpus/round7_launch.sh @@ -0,0 +1,31 @@ +#!/usr/bin/env bash +# round7_launch.sh -- start the round-7 cold baseline detached, with the VRAM +# sampler beside it, after the smoke test has proved the leg scores. +# +# Operational copy lives at /mnt-1/benchmarks/round7_launch.sh. +# +# Refuses to launch unless round7_smoke.sh has left a Tier B result on this +# pin: a Tier-B-only failure is the exact 2026-09-04 defect (#2971), and the +# sweep would still write MODEL_DONE over the hole. +set -u +BASE=${BASE:-/mnt-1/benchmarks} +SMOKE=${SMOKE:-$BASE/smoke-round7/tierB_smoke.json} +[ -s "$SMOKE" ] || { echo "ABORT: no Tier B smoke result at $SMOKE -- run round7_smoke.sh first"; exit 1; } +python3 - "$SMOKE" <<'PY' || exit 1 +import json, sys +d = json.load(open(sys.argv[1])) +assert d["case_count"] == 17, f"smoke scored {d['case_count']} cases, not 17 -- wrong pin or corpus" +# 83, not the 79 the resume plan guessed: max per case is required_groups + 1, +# measured on this pin by the smoke run (70/83 A, 69/83 B for qwen2.5:7b). +assert d["total_max_score"] == 83, f"smoke max is {d['total_max_score']}, not 83" +assert d["total_score"] > 0, "smoke scored 0 -- empty answers, do not launch" +print(f"smoke ok: {d['total_score']}/{d['total_max_score']} over {d['case_count']} cases") +PY +if pgrep -f "sweep_extra[.]sh" >/dev/null 2>&1 || pgrep -f "record_baseline[.]py" >/dev/null 2>&1; then + echo "ABORT: a sweep is already running"; exit 1 +fi +cd "$BASE" || exit 1 +pgrep -f "keep_and_sample[.]sh" >/dev/null 2>&1 || \ + { setsid nohup bash "$BASE/keep_and_sample.sh" >> "$BASE/keepsample.log" 2>&1 < /dev/null & echo "sampler started"; } +setsid nohup bash "$BASE/round7_coldrun.sh" >> "$BASE/round7.log" 2>&1 < /dev/null & +echo "launched round7_coldrun.sh -> $BASE/round7.log" diff --git a/analysis/ghidra/benchmarks/corpus/round7_prep_pin.sh b/analysis/ghidra/benchmarks/corpus/round7_prep_pin.sh new file mode 100755 index 000000000..f97ac3fee --- /dev/null +++ b/analysis/ghidra/benchmarks/corpus/round7_prep_pin.sh @@ -0,0 +1,30 @@ +#!/usr/bin/env bash +# prep_round7_clone.sh -- pinned checkout for round 7 (epic #3079): a SECOND +# clone, detached at the round-7 pin. The a99e765 clone stays untouched -- its +# untracked transcripts are the phase-1/2 evidence and resume_phases.sh guards +# that HEAD. Operational copy: /mnt-1/benchmarks/round7_prep_pin.sh +set -euo pipefail +PIN=${PIN:-32dbdeb1face8c8e4791d31a8f4fbbe321e4f6fa} +DST=${DST:-/mnt-1/benchmarks/APIARY-round7} +OLD=${OLD:-/mnt-1/benchmarks/APIARY} +url=$(git -C "$OLD" remote get-url origin) +if [ -d "$DST/.git" ]; then + echo "clone exists: $DST" +else + git clone --quiet --no-checkout "$url" "$DST" +fi +git -C "$DST" fetch --quiet origin "$PIN" 2>/dev/null || git -C "$DST" fetch --quiet origin +git -C "$DST" checkout --quiet --detach "$PIN" +echo "round7 clone HEAD: $(git -C "$DST" rev-parse HEAD)" +python3 - "$DST" <<'PY' +import json, sys +from pathlib import Path +d = Path(sys.argv[1]) / "analysis/ghidra/benchmarks/corpus" +r = json.load(open(d / "rev_cases_v2_rubric.json")) +cases = r.get("cases", r) +m = json.load(open(d / "manifest.json")) +print("rubric cases:", len(cases)) +print("manifest builds:", len(m["builds"])) +sl = [b for b in m["builds"] if b.get("toolchain") == "gcc-x86_64" and b.get("opt_level") == "-O0"] +print("gcc-x86_64 -O0 builds:", len(sl)) +PY diff --git a/analysis/ghidra/benchmarks/corpus/round7_prepare.sh b/analysis/ghidra/benchmarks/corpus/round7_prepare.sh new file mode 100644 index 000000000..5f667edd2 --- /dev/null +++ b/analysis/ghidra/benchmarks/corpus/round7_prepare.sh @@ -0,0 +1,25 @@ +#!/usr/bin/env bash +# round7_prepare.sh -- everything round 7 needs before its first GPU minute, +# in dependency order, each step resume-safe: +# 1. round7_prep_pin.sh pinned checkout, detached at the round-7 pin +# 2. round7_build_corpus.sh 17-case corpus rebuilt in the provenance container, +# verified byte-for-byte against the pinned manifest +# 3. round7_cache.sh 17-case Tier B (Ghidra) cache in its own directory +# Then run round7_smoke.sh, read its two lines, and only then round7_launch.sh. +# +# Operational copy lives at /mnt-1/benchmarks/round7_prepare.sh. +set -euo pipefail +BASE=${BASE:-/mnt-1/benchmarks} +log() { echo "$(date -u +%FT%TZ) $*"; } +log "step 1/3 pin" +bash "$BASE/round7_prep_pin.sh" +if [ -f "$BASE/corpus-round7/manifest.json" ] && diff -q "$BASE/corpus-round7/manifest.json" \ + "$BASE/APIARY-round7/analysis/ghidra/benchmarks/corpus/manifest.json" >/dev/null 2>&1; then + log "step 2/3 corpus already built and identical to the pinned manifest -- skipping" +else + log "step 2/3 corpus (apt + 850 builds in debian:trixie-slim)" + bash "$BASE/round7_build_corpus.sh" +fi +log "step 3/3 Tier B cache" +bash "$BASE/round7_cache.sh" +log "ROUND7_PREPARED -- next: bash $BASE/round7_smoke.sh" diff --git a/analysis/ghidra/benchmarks/corpus/round7_smoke.sh b/analysis/ghidra/benchmarks/corpus/round7_smoke.sh new file mode 100755 index 000000000..2b16c572b --- /dev/null +++ b/analysis/ghidra/benchmarks/corpus/round7_smoke.sh @@ -0,0 +1,34 @@ +#!/usr/bin/env bash +# round7_smoke.sh -- prove the round-7 benchmark LEG scores on the new pin +# before the roster is launched (#1947 rule 7: verify the benchmark leg, not +# the pull). One Tier A and one Tier B run of a small local model into a +# separate directory that no results glob reads. +# +# Operational copy lives at /mnt-1/benchmarks/round7_smoke.sh. +set -u +BASE=${BASE:-/mnt-1/benchmarks} +REPO=${REPO:-$BASE/APIARY-round7} +OUT=${OUT:-$BASE/smoke-round7} +CACHE=${CACHE:-$BASE/tierb-cache-round7} +TAG=${TAG:-qwen2.5:7b-instruct-q4_K_M} +mkdir -p "$OUT" +cd "$REPO" || exit 1 +for tier in A B; do + extra=""; [ "$tier" = "B" ] && extra="--ghidra-cache $CACHE" + docker exec ghidra-ollama-1 ollama stop "$TAG" >/dev/null 2>&1 + timeout 3600 python3 analysis/ghidra/benchmarks/corpus/record_baseline.py \ + --tier "$tier" $extra --model "$TAG" --operator smoke-round7 --provenance synthetic \ + --output "$OUT/tier${tier}_smoke.json" > "$OUT/tier${tier}.log" 2>&1 + rc=$? + summary=$(python3 - "$OUT/tier${tier}_smoke.json" <<'PY' 2>&1 +import json, sys +d = json.load(open(sys.argv[1])) +cases = d["cases"] +empty = sum(1 for c in cases.values() if c.get("empty_answer")) +print(f"score {d['total_score']}/{d['total_max_score']} cases {d['case_count']} empty_answers {empty} digest {d.get('model_digest','?')[:12]}") +PY +) + echo "$(date -u +%FT%TZ) tier $tier rc=$rc $summary" +done +docker exec ghidra-ollama-1 ollama stop "$TAG" >/dev/null 2>&1 +echo "transcript runs in the round-7 clone: $(ls "$REPO/docs/benchmarks/runs" 2>/dev/null | wc -l)" diff --git a/analysis/ghidra/benchmarks/corpus/sweep_extra.sh b/analysis/ghidra/benchmarks/corpus/sweep_extra.sh index 2c64bea7e..37c9c922c 100755 --- a/analysis/ghidra/benchmarks/corpus/sweep_extra.sh +++ b/analysis/ghidra/benchmarks/corpus/sweep_extra.sh @@ -60,6 +60,12 @@ LIVE_WORKERS=${LIVE_WORKERS:-"hp-llm-worker ghidra-revdeck-1"} # still protects the filesystem that holds the Docker volumes and the ES data. KEEP_WEIGHTS_ABOVE_GB=${KEEP_WEIGHTS_ABOVE_GB:-1000} +# #3087: round 7 scores on its own pin with its own 17-case Tier B cache and +# its own operator tag, so both are overridable; the defaults are the a99e765 +# sweep's, unchanged. round7_coldrun.sh sets them. +GHIDRA_CACHE=${GHIDRA_CACHE:-/mnt-1/benchmarks/tierb-cache} +OPERATOR=${OPERATOR:-bg-1947extra} + # #2738: fail fast on any roster entry Ollama's client-side hf.co name # validation would reject before a sweep wastes time discovering it -- # see /mnt-1/benchmarks/oversized-model-aliases.tsv for the bisection and @@ -151,7 +157,7 @@ do_run() { # tier slug tag n local tier="$1" slug="$2" tag="$3" n="$4" local out="$BASE/tier${tier}_${slug}_run${n}.json" [ -f "$out" ] && { echo "$(date -u +%H:%M:%S) skip $tier $slug run$n"; return 0; } - local extra=""; [ "$tier" = "B" ] && extra="--ghidra-cache /mnt-1/benchmarks/tierb-cache" + local extra=""; [ "$tier" = "B" ] && extra="--ghidra-cache $GHIDRA_CACHE" local try=1 while [ $try -le $MAXTRY ]; do docker exec ghidra-ollama-1 ollama stop "$tag" >/dev/null 2>&1 @@ -159,7 +165,7 @@ do_run() { # tier slug tag n echo "$(date -u +%H:%M:%S) start $tier $slug run$n try$try" timeout 10800 python3 analysis/ghidra/benchmarks/corpus/record_baseline.py \ --tier "$tier" $extra --model "$tag" \ - --operator bg-1947extra --provenance synthetic \ + --operator "$OPERATOR" --provenance synthetic \ --output "$out" > "$BASE/logs/x_tier${tier}_${slug}_run${n}_try${try}.log" 2>&1 local rc=$? if [ $rc -eq 0 ] && [ -f "$out" ]; then diff --git a/docs/benchmarks/plans/2026-09-06-round7-hermes-init.md b/docs/benchmarks/plans/2026-09-06-round7-hermes-init.md new file mode 100644 index 000000000..56c0dee6d --- /dev/null +++ b/docs/benchmarks/plans/2026-09-06-round7-hermes-init.md @@ -0,0 +1,142 @@ +# init.md — round 7 handoff for the orchestrator (Hermes → Claude Code CLI) + +**Read this first.** It names the plan of record, the state of the GPU, the +kickoff order, the dispatch pattern and the hard rules. Written 2026-09-06, +after the round-7 cold baseline was launched. Epic **#3079**, children +**#3080–#3088**. + +## 1. Mode + +You are the **orchestrator**. You do not author code, tests, scripts, compose +files, reviews or designs — you dispatch them to the Claude Code CLI and you +bookkeep. Your own outputs are: preflight reports, dispatch prompts (problem +statements, never solution specs), commits/pushes/PRs of dispatched work, issue +comments, ledger entries. When a real decision appears (secrets, budget, a +merge on a contested change, parking an issue), stop and ask Xore with concrete +options and a recommendation. + +Start in **PREP mode**: verify the state below, fix environment gaps by probing, +report readiness, and **wait for Xore to say START** before the first coder +session. + +## 2. The plan of record and where things live + +| what | where | +|---|---| +| plan | `docs/benchmarks/plans/2026-09-06-round7-unsloth-train-requant-ollama.md` — merged to `main` by PR #3089 | +| working conventions | plan §15 (grit, rtk, gh, two-agent cap) and this file §5 | +| live state of the benchmark host | `homeserver:/mnt-1/benchmarks/STATE-2026-09-06-round7-fold.md` | +| coding tree | the workstation checkout `/home/xore/Github/APIARY` (grit-initialised, 101k symbols). **Never code in the homeserver's pinned clones** (`/mnt-1/benchmarks/APIARY` at a99e765, `/mnt-1/benchmarks/APIARY-round7` at 32dbdeb1): they are scoring vintages, detached on purpose, and must not move or be committed into. Operational copies of scripts are `scp`'d to `/mnt-1/benchmarks/` exactly as every `round7_*.sh` was | +| training work area (to be created by #3080) | `homeserver:/mnt-1/training/` — corpus, calibration sets, runs, pools; `/mnt-1/hf-cache` | + +## 3. GPU gate — the one card is taken until further notice + +The round-7 cold baseline is running: `round7_coldrun.sh` → `sweep_extra.sh`, +pin `32dbdeb1`, 96 tags, cold, 17 cases / 83, results `/mnt-1/benchmarks/round7/`. +Started 2026-09-06 14:08Z; expect 2–4 days. + +```bash +ssh homeserver 'pgrep -af "round7_coldrun|coldrun.sh|sweep_extra.sh|record_baseline.py|requant_sweep.sh|chain_"; tail -5 /mnt-1/benchmarks/round7.log' +``` + +Alive processes = the GPU is taken. **Never stop, restart or share it.** An +empty `nvidia-smi` between legs is not idle. The card is free only on the +positive condition: every tag in `models_round7.txt` has both tier files or an +`UNMEASURED` marker in `round7/`, **and** no process above is alive. #3087's +`chain_round7.sh` encodes that condition; until it exists, check by hand. + +Until the card frees, only the **no-GPU** legs run. + +## 4. Kickoff order + +``` +PREP (you) gate check ▸ issue/PR recon ▸ grit gc + grit status on the workstation ▸ probe homeserver gaps + (/mnt-1/training missing, HF cache location, no OpenRouter key yet) ▸ report ▸ WAIT FOR START +wave 1 no GPU, 2 sessions #3080 toolchain (container + export_to_ollama.sh, NO smoke yet) + #3082 corpus v1 (slices S3/S4/S5/S6 + decontaminate.py; S1/S2 local-teacher labels wait for the card) +wave 2 no GPU, 2 sessions #3081 merge + convert of the REx86 adapter (CPU) · #3086 calibration sets + #3087 slots_sweep.sh + chain_round7.sh + pooled-claims rescoring script +card frees #3080 smoke ▸ #3082 local-teacher labels ▸ #3081 scoring ▸ #3086 imatrix + ladders + ▸ #3083 CPT ▸ #3084 SFT ▸ #3085 DPO → GRPO ▸ #3087 scores every artefact as it lands +gated #3088 only after #3084's first result and Xore's call +``` + +Each wave: ALL coding (≤ 2 parallel) → ALL reviews (strictly sequential) → +ALL fixes (≤ 2 parallel) → ALL ship (sequential, push and exit). Never +interleave stages. Fix iterations hard cap 4, then escalate to Xore. + +## 5. Tooling every session uses — grit, ponytail, rtk, gh + +- **grit** (mandatory, function-level locks, replaces bare worktrees). + Orchestrator claims **before** dispatching: + `grit gc` → `grit symbols --file 'analysis/ghidra/benchmarks/corpus/*'` (or + `analysis/ghidra/training/*`; "No symbols found" for a directory that does not + exist yet is normal — verify with `grit status`) → `grit plan -a issue- -i ""` + → `grit claim -a issue--coder -i "" :: …` + (`--with-deps`, `--queue`) → dispatch the coder with `.grit/worktrees/issue--coder/` + as cwd → tester gets `grit claim -a issue--tester --mode read …` → + `grit heartbeat -a … --ttl 900` on long legs → after review, `grit done -a issue--coder` + (**sequential, never two at once**) → `grit init` after the merge when files were + added. One agent name per session, never shared. Absolute paths inside grit worktrees. +- **ponytail** (mandatory with grit): every coder session runs under `/ponytail` + (default `full`; `/ponytail ultra` for sprawl-prone areas — the training + toolchain and corpus builders are exactly that); `/ponytail-review` on the diff + before ship. grit prevents collisions, ponytail prevents over-build. +- **rtk**: every shell command goes through the hook's `rtk` rewrite; never + bypass it; `rtk proxy ` only for genuinely raw output; `rtk gain --history` + in every closing report. +- **gh**: `gh issue view ` **and** `gh pr list --search ""` before claiming; + `in-progress` label + assignee Xore + comment; status comment at each milestone; + push → PR → exit (no CI polling as a standing activity); merge only on + `mergeStateStatus == CLEAN` with `gh pr merge --squash --delete-branch`; + squash drops `Closes #N` — audit and close stragglers by hand with an evidence + comment; no AI tooling or model names in commits, PRs, comments or issues; no + credentials or production domains in issues; times in Europe/Berlin. +- **Dispatch pattern**: write the prompt file first, then + `{ cat STAGE-.md; printf '\nArguments: githubissuesN \n'; } | claude -p --model --effort --output-format json --permission-mode …`. + Model ownership: initial code → Sonnet 5 high; reviews and remediations → Opus 5 MAX; + CI fixes → Opus 5 high→max; design content → Fable on high. No `--max-turns` caps; + budget is the only cap. On a 429: wait, probe (`claude -p "Reply with: ok"`), `--resume`; + never switch models. On session-limit failures: probe, wait for reset, continue + with a continuation prompt that names the existing STAGE files. +- **Runner fleet**: 2 online max (`supermicro`, `supermicro-ci-2`); the rest stay disabled. + +## 6. Hard rules for every coder prompt (verbatim into each STAGE file) + +- The 17 benchmark corpus programs are the test set. Never use them, their + decompiled text, rubric terms, claim pool or transcripts for training, + calibration or RL prompts. #3082's decontamination report (0 hits) must exist + before any round-7 score is quoted. +- Captured honeypot data never leaves the homeserver: not git, not Hugging Face, + not an API teacher. Local teachers only for captured slices. OpenRouter + open-weight teachers only for synthetic slices; the key lives in a 0600 file on + the host, never in the repo or an issue. +- Training container: pinned by digest, GPU `GPU-18a00c7e-670a-c305-a2aa-20e3a71917a3` + only (never `--gpus all`), no restart policy, exits between runs, no docker socket. + Every chain waits on a positive condition, never on `nvidia-smi` looking idle. +- Export path: `merged_16bit` → `convert_hf_to_gguf.py` in `ghcr.io/ggml-org/llama.cpp:full` + → `llama-imatrix` + `llama-quantize` → `ollama create` with the base's + TEMPLATE/PARAMETERs copied from `ollama show --modelfile` and diffed against + Unsloth's generated Modelfile. Never `merged_4bit`, never `ollama create --quantize`. +- One scoring pin for the whole round (`32dbdeb1`, 17 cases / **83**, pooled claims, + sessions and Rev·Deck slots), cold protocol, `STOP_WORKERS=1`, N=2 → 3 → 5, + `UNMEASURED` / `UNMEASURABLE` markers never zeros, transcripts committed for + synthetic runs. Every trained row vs its own base at the same quant **and** vs + `qwen3:14b`; every requant row vs the as-published quant and phase 3's plain level. +- Commit every script, config, manifest, run card and decontamination report; + weights and adapters stay on `/mnt-1/training` and get mirrored. Nothing lives + only in `/mnt-1`. Operational copies carry the "operational copy lives at …" header. +- Two sessions max, two grit names, only one may hold the GPU. + +## 7. Definition of done, per issue + +The issue's own acceptance checklist. The closing comment carries: what landed +(paths, PR numbers), what was measured (numbers with their pin), what is still +open and why, and `rtk gain --history`. + +## 8. What Xore still owes the round (ask, do not guess) + +1. OpenRouter key on the homeserver (0600 file) — needed by #3082's S3 labels only. +2. 2 × 16 GiB DDR4 registered ECC RDIMM (`DIMME1` + `DIMMF1`) — not a blocker. +3. Private Hugging Face repos for adapters — default off. +4. The word **START**. diff --git a/docs/benchmarks/plans/2026-09-06-round7-unsloth-train-requant-ollama.md b/docs/benchmarks/plans/2026-09-06-round7-unsloth-train-requant-ollama.md index da06358d7..8abc583c2 100644 --- a/docs/benchmarks/plans/2026-09-06-round7-unsloth-train-requant-ollama.md +++ b/docs/benchmarks/plans/2026-09-06-round7-unsloth-train-requant-ollama.md @@ -14,7 +14,30 @@ calibration-set hashes, decontamination reports. --- -## 1. Where we are — measured, not assumed (2026-09-06) +## 0. What changed after this plan was first written (2026-09-06, afternoon) + +The operator folded the #1947 cold re-run into round 7 instead of letting it +spend 2–4 GPU-days on the saturated 14-case pin. Done the same day, all of it +committed as `analysis/ghidra/benchmarks/corpus/round7_*.sh` and recorded in +`/mnt-1/benchmarks/STATE-2026-09-06-round7-fold.md`: + +| step | result | +|---|---| +| old-pin cold re-run stopped | after 13 models / 52 result files in `1947cold/` — valid cold cells on the OLD pin, kept, marked `ABORTED-2026-09-06.txt`, never extended | +| phase 3 (#2245) | had completed at 10:26Z; `ornith-35b-selfquant:Q4_K_M` scored **0 on all four runs** = UNMEASURABLE on that pin, re-measured in round 7 | +| round-7 pin | second clone `APIARY-round7`, detached at **`32dbdeb1`** — the a99e765 clone untouched, its transcripts mirrored first | +| corpus | 17-case corpus rebuilt in the provenance container, byte-identical to the pinned manifest → `corpus-round7/` | +| Tier B cache | `tierb-cache-round7/`, 17 entries, 0 errors, Ghidra 11.3.2 | +| smoke | `qwen2.5:7b-instruct-q4_K_M`: **Tier A 70/83, Tier B 69/83**, 17 cases, 0 empty answers — **the max is 83, not 79** | +| **launched 14:08Z** | `round7_coldrun.sh` → `sweep_extra.sh`, 96 tags, cold, N=2 → 3 → 5, results `round7/`, log `round7.log`, VRAM sampler beside it. Expect 2–4 days | + +So the "GPU frees" line in §9 now means **"when `round7_coldrun.sh` completes"**, +and the round-7 ghidra-slot baseline is no longer future work for #3087 — it +is running. What #3087 still owns: the sessions/Rev·Deck legs +(`slots_sweep.sh`), pooled-claims rescoring from the transcripts, +`chain_round7.sh` for the training legs, and the write-up. + +## 1. Where we are — measured, not assumed (2026-09-06, morning) ### 1.1 The #1947 sweep is finishing @@ -85,7 +108,7 @@ Neither is fixed by pulling one more tag. Both are the subject of this round. | "bigger models" | requant the 27–35 B class to GPU-resident **and** the 100 B+ RAM-offload class (123 B / 218 B) | §7 R2; GLM-4.6 357 B stays a measured rejection | | training data | a separate, decontaminated corpus; real captured ES data allowed as input, never leaves the host; captured slices use a **local teacher**; synthetic slices may use **open-weight frontier teachers via OpenRouter** | §6 | | compute | **local card only**, queued behind the cold re-run; above 14 B only at the VRAM edge (#3088, gated) | §9 | -| harness for round 7 | **main's 17-case / 79 rubric + pooled claims + all three slots, one new pin** | §8; `a99e765` numbers become historical context | +| harness for round 7 | **main's 17-case / 83 rubric + pooled claims + all three slots, one new pin** | §8; `a99e765` numbers become historical context | --- @@ -276,7 +299,9 @@ quant and against phase 3's plain level at matched size (isolates the recipe). - **Pin.** `main` at the commit the round starts on; recorded in every result and the write-up; never moved mid-round. `resume_phases.sh`-style guard. -- **Ghidra slot.** `record_baseline.py --tier A` / `--tier B`, 17 cases / 79, +- **Ghidra slot.** `record_baseline.py --tier A` / `--tier B`, 17 cases / **83** + (max per case is `required_groups + 1`; the resume plan's "79" was a guess — + the smoke run on this pin measured 70/83 A, 69/83 B for `qwen2.5:7b`), `tierb-cache` regenerated for 17 cases on the current `ghidra-ghidra-1` (`GHIDRA_VERSION` exported — #2983), injection gate v3 paired verdicts, **plus** pooled-claims scoring with `claims.py` against `tier-a-v1.json` @@ -432,10 +457,20 @@ this round's control), #1804, #2279 / #2985, #356 / #1523, #2969, #3023. ```bash # is the card free, and who holds it ssh homeserver 'nvidia-smi --query-compute-apps=pid,used_memory --format=csv; \ - pgrep -af "coldrun.sh|sweep_extra.sh|record_baseline.py|requant_sweep.sh|chain_"' + pgrep -af "round7_coldrun|coldrun.sh|sweep_extra.sh|record_baseline.py|requant_sweep.sh|chain_"' + +# state of the round-7 cold baseline (must be complete before any training leg) +ssh homeserver 'tail -20 /mnt-1/benchmarks/round7.log' # must show BOTH "done A" and "done B" lines +ssh homeserver 'ls /mnt-1/benchmarks/round7/tierA_*.json | wc -l; ls /mnt-1/benchmarks/round7/tierB_*.json | wc -l' +# 96 tags x 2 tiers x N>=2; complete = every roster tag has both tier files or an UNMEASURED marker +# stop, in order (sweep_extra.sh TRAPS TERM and keeps going -- it needs KILL): +ssh homeserver 'pkill -TERM -f "bash /mnt-1/benchmarks/round7_coldrun[.]sh"; \ + pkill -KILL -f "bash /mnt-1/benchmarks/sweep_extra[.]sh"; pkill -TERM -f "record_baseline[.]py"; \ + docker start ghidra-revdeck-1' # then quarantine any partial result in round7/ and ollama stop the tag -# state of the cold re-run (must be complete before any training leg) -ssh homeserver 'tail -5 /mnt-1/benchmarks/coldchain.log; ls /mnt-1/benchmarks/1947cold/ | wc -l' +# prepare the pin again from scratch (resume-safe; clone -> corpus -> Tier B cache), then smoke, then launch +ssh homeserver 'bash /mnt-1/benchmarks/round7_prepare.sh && bash /mnt-1/benchmarks/round7_smoke.sh' +ssh homeserver 'bash /mnt-1/benchmarks/round7_launch.sh' # training work area layout (created by #3080) # /mnt-1/training/corpus-v1/ /mnt-1/training/calib/ /mnt-1/training/runs// @@ -487,7 +522,7 @@ grit done -a issue--coder # auto-commit, rebase, se - New files have no symbols yet: claim the neighbours you touch (the compose file, the README, the sibling script), and **re-run `grit init` after the merge** so the new symbols are indexed for the next agent. -- Relative-path dependencies (`path = "../x"`, `-r ../requirements.txt`) break +- Relative-path dependencies (`path = "../"`, `-r ../`) break inside `.grit/worktrees/`; use absolute paths. - After any large merge to `main`, `grit init` again. diff --git a/scripts/doc-path-lint-allowlist.txt b/scripts/doc-path-lint-allowlist.txt index 53c05edba..7e75d7dbb 100644 --- a/scripts/doc-path-lint-allowlist.txt +++ b/scripts/doc-path-lint-allowlist.txt @@ -82,3 +82,8 @@ tests/test_importer.py::ScanSourceSameMtimeRaceTest # era class reference; same tests/test_es_consume.py # root-relative test path; the consuming doc predates the per-component split tests/test_claims.py # root-relative test path in docs/benchmarks/claim-pools; actual at analysis/ghidra/benchmarks/tests/test_claims.py tests/test_claims.py::ReviewQueueFreshnessTest # era class reference; same source as above + +# --- round-7 training tree, created by #3080 (epic #3079); the plan doc +# names where the toolchain record and the scripts will land --- +analysis/ghidra/training # training tree the round-7 plan creates in #3080; not tracked yet +analysis/ghidra/training/TOOLCHAIN.md # resolved-version record #3080 writes into that tree; not tracked yet