diff --git a/EVIDENCE.md b/EVIDENCE.md index 17d49104a..7f8f20a8b 100644 --- a/EVIDENCE.md +++ b/EVIDENCE.md @@ -1,5 +1,9 @@ # #2646 — warm-slot reproducibility on the production triage path +> `/mnt-1` paths below are dead as of #3158/#3159 (decommissioned 2026-09-09). Left +> as-written since this is a dated record; see `docs/HOMESERVER-DISK-LAYOUT.md` for +> current layout. + ## 0. What this session could and could not do State this first so nothing below is read as stronger than it is. diff --git a/analysis/ghidra/benchmarks/corpus/chain_cold.sh b/analysis/ghidra/benchmarks/corpus/chain_cold.sh index cf213f84c..601e0f2a6 100755 --- a/analysis/ghidra/benchmarks/corpus/chain_cold.sh +++ b/analysis/ghidra/benchmarks/corpus/chain_cold.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash # chain_cold.sh -- start the full cold re-run once phase 3's scoring is done. # -# Operational copy lives at /mnt-1/benchmarks/chain_cold.sh. +# Operational copy lives at /var/benchmarks/chain_cold.sh. # # --------------------------------------------------------------------------- # Why the wait condition is positive, not "is the GPU idle" @@ -20,7 +20,7 @@ # is a wasted day of GPU rather than an error message. set -u -BASE=${BASE:-/mnt-1/benchmarks} +BASE=${BASE:-/var/benchmarks} RESULTS=${RESULTS:-$BASE/1947full} TAGLIST=${TAGLIST:-$BASE/models_requant.txt} MAX_WAIT_MIN=${MAX_WAIT_MIN:-4320} # 72h diff --git a/analysis/ghidra/benchmarks/corpus/chain_phase3.sh b/analysis/ghidra/benchmarks/corpus/chain_phase3.sh index dd92c2104..c225b9435 100755 --- a/analysis/ghidra/benchmarks/corpus/chain_phase3.sh +++ b/analysis/ghidra/benchmarks/corpus/chain_phase3.sh @@ -2,7 +2,7 @@ # chain_phase3.sh -- run phase 3's SCORING half once phase 2 has actually # finished with the GPU. # -# Operational copy lives at /mnt-1/benchmarks/chain_phase3.sh. Committed here +# Operational copy lives at /var/benchmarks/chain_phase3.sh. Committed here # because the previous generation of chain scripts (chain2b.sh, chain3.sh) were # never committed, did not survive the rebuild, and took phases 2.5/3/5 with # them (#2985). @@ -26,7 +26,7 @@ # green light to spend GPU hours on the next phase. set -u -BASE=${BASE:-/mnt-1/benchmarks} +BASE=${BASE:-/var/benchmarks} RESULTS=${RESULTS:-$BASE/1947full} ROSTER=${ROSTER:-$BASE/models_extra_all.txt} TAGLIST=${TAGLIST:-$BASE/models_requant.txt} diff --git a/analysis/ghidra/benchmarks/corpus/chain_round7.sh b/analysis/ghidra/benchmarks/corpus/chain_round7.sh index ddb5545bb..e02f6a363 100755 --- a/analysis/ghidra/benchmarks/corpus/chain_round7.sh +++ b/analysis/ghidra/benchmarks/corpus/chain_round7.sh @@ -2,7 +2,7 @@ # chain_round7.sh -- #3087: fire the round-7 slot sweeps once every round-7 # roster tag is measured on the ghidra slot AND the GPU work area is idle. # -# Operational copy: /mnt-1/benchmarks/chain_round7.sh. +# Operational copy: /var/benchmarks/chain_round7.sh. # # --------------------------------------------------------------------------- # Why the wait condition is POSITIVE, not "is the GPU idle" -- the same @@ -16,7 +16,7 @@ # every tag in models_round7.txt has both tier files # (tierA__run1.json + tierB__run1.json) OR an # UNMEASURED_slots.../UNMEASURED_.status marker in -# /mnt-1/benchmarks/round7/ +# /var/benchmarks/round7/ # # ...AND no sweep_extra.sh / round7_coldrun.sh's own drivers # (coldrun.sh) / record_baseline.py process is alive. Both must hold at once: @@ -25,7 +25,7 @@ # as the second line of defence. set -u -BASE=${BASE:-/mnt-1/benchmarks} +BASE=${BASE:-/var/benchmarks} RESULTS=${RESULTS:-$BASE/round7} TAGLIST=${TAGLIST:-$BASE/models_round7.txt} MAX_WAIT_MIN=${MAX_WAIT_MIN:-10080} # 7 days: the cold run alone is 2-4 diff --git a/analysis/ghidra/benchmarks/corpus/coldprobe.sh b/analysis/ghidra/benchmarks/corpus/coldprobe.sh index 0a4114e7b..06d996509 100755 --- a/analysis/ghidra/benchmarks/corpus/coldprobe.sh +++ b/analysis/ghidra/benchmarks/corpus/coldprobe.sh @@ -28,9 +28,9 @@ # Preconditions this asserts rather than assumes: hp-llm-worker down, # sweep_extra.sh not running, no record_baseline in flight. set -u -REPO=/mnt-1/benchmarks/APIARY -OUT=/mnt-1/benchmarks/coldprobe -CACHE=/mnt-1/benchmarks/tierb-cache +REPO=/var/benchmarks/APIARY +OUT=/var/benchmarks/coldprobe +CACHE=/var/benchmarks/tierb-cache mkdir -p "$OUT/logs" die() { echo "ABORT: $*" >&2; exit 1; } diff --git a/analysis/ghidra/benchmarks/corpus/coldrun.sh b/analysis/ghidra/benchmarks/corpus/coldrun.sh index 43110c8e6..4d3756189 100755 --- a/analysis/ghidra/benchmarks/corpus/coldrun.sh +++ b/analysis/ghidra/benchmarks/corpus/coldrun.sh @@ -2,7 +2,7 @@ # coldrun.sh -- re-measure the WHOLE #1947 roster under one uniform regime: # cold slot, N=2 with automatic escalation, live workers stopped. # -# Operational copy lives at /mnt-1/benchmarks/coldrun.sh. +# Operational copy lives at /var/benchmarks/coldrun.sh. # # --------------------------------------------------------------------------- # Why this exists @@ -37,7 +37,7 @@ # it away to save disk would be the same mistake as deleting the weights. set -u -BASE=${BASE:-/mnt-1/benchmarks} +BASE=${BASE:-/var/benchmarks} COLD=${COLD:-$BASE/1947cold} ROSTER=${ROSTER:-$BASE/models_cold_all.txt} REPO=${REPO:-$BASE/APIARY} diff --git a/analysis/ghidra/benchmarks/corpus/gptoss_finalize.sh b/analysis/ghidra/benchmarks/corpus/gptoss_finalize.sh index 114937fd4..f4ce70213 100755 --- a/analysis/ghidra/benchmarks/corpus/gptoss_finalize.sh +++ b/analysis/ghidra/benchmarks/corpus/gptoss_finalize.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash # gptoss_finalize.sh -- #1947 phase 2.5 / #2279 step 3. # -# Operational copy lives at /mnt-1/benchmarks/gptoss_finalize.sh. Committed here +# Operational copy lives at /var/benchmarks/gptoss_finalize.sh. Committed here # because its predecessor (gptoss_rerun.sh) was never committed and did not # survive the 2026-09-03/04 rebuild (#2985). # @@ -40,7 +40,7 @@ # DRY_RUN=1 bash gptoss_finalize.sh # report only set -u -BASE=${BASE:-/mnt-1/benchmarks} +BASE=${BASE:-/var/benchmarks} RESULTS=${RESULTS:-$BASE/1947full} DRY_RUN=${DRY_RUN:-0} diff --git a/analysis/ghidra/benchmarks/corpus/keep_and_sample.sh b/analysis/ghidra/benchmarks/corpus/keep_and_sample.sh index e758849f0..cfede1960 100755 --- a/analysis/ghidra/benchmarks/corpus/keep_and_sample.sh +++ b/analysis/ghidra/benchmarks/corpus/keep_and_sample.sh @@ -3,7 +3,7 @@ # # Two jobs, both read-mostly: # -# 1. RE-CREATE THE LOST SAMPLER (#2245). /mnt-1/benchmarks/vram_samples.tsv was +# 1. RE-CREATE THE LOST SAMPLER (#2245). /var/benchmarks/vram_samples.tsv was # the empirical record of served size and CPU/GPU split per model -- the input # #2245's category-1 "which models actually spill" list was supposed to rest # on. It was wiped with the rest of the work area (#2971) and is in no mirror, @@ -26,8 +26,8 @@ # Stop with: pkill -f keep_and_sample.sh (leaves every alias in place) set -u OLLAMA=ghidra-ollama-1 -SAMPLES=/mnt-1/benchmarks/vram_samples.tsv -KEPT=/mnt-1/benchmarks/kept-aliases.tsv +SAMPLES=/var/benchmarks/vram_samples.tsv +KEPT=/var/benchmarks/kept-aliases.tsv INTERVAL=60 oll() { docker exec "$OLLAMA" ollama "$@" 2>/dev/null; } diff --git a/analysis/ghidra/benchmarks/corpus/probe_at_gap.sh b/analysis/ghidra/benchmarks/corpus/probe_at_gap.sh index 43d63a4c2..68aa4d888 100755 --- a/analysis/ghidra/benchmarks/corpus/probe_at_gap.sh +++ b/analysis/ghidra/benchmarks/corpus/probe_at_gap.sh @@ -13,7 +13,7 @@ # every model that already has both tier run1 files, so the relaunch resumes # exactly where it stopped. set -u -BASE=/mnt-1/benchmarks +BASE=/var/benchmarks LOG=$BASE/probe_at_gap.log say() { echo "$(date -u +%FT%TZ) $*" | tee -a "$LOG"; } diff --git a/analysis/ghidra/benchmarks/corpus/recreate-local-tags.sh b/analysis/ghidra/benchmarks/corpus/recreate-local-tags.sh index 71b91c5f4..a19b1159e 100755 --- a/analysis/ghidra/benchmarks/corpus/recreate-local-tags.sh +++ b/analysis/ghidra/benchmarks/corpus/recreate-local-tags.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash # recreate-local-tags.sh -- #2985: lost in the same rebuild as # gptoss_rerun.sh/requant_sweep.sh/slots_sweep.sh, and never committed before -# now. Operational copy lives at /mnt-1/benchmarks/recreate-local-tags.sh. +# now. Operational copy lives at /var/benchmarks/recreate-local-tags.sh. # # #2695: rebuilds Ollama-local tags that models_extra_all.txt references but # that sweep_extra.sh cannot `ollama pull` (they're not registry tags). diff --git a/analysis/ghidra/benchmarks/corpus/requant_sweep.sh b/analysis/ghidra/benchmarks/corpus/requant_sweep.sh index 8d33f09bb..fb5730c97 100755 --- a/analysis/ghidra/benchmarks/corpus/requant_sweep.sh +++ b/analysis/ghidra/benchmarks/corpus/requant_sweep.sh @@ -3,7 +3,7 @@ # from an f16 master and measure every level with the same scorer, at the same # pin, as the as-published rows it is being compared against. # -# Operational copy runs from /mnt-1/benchmarks/requant_sweep.sh on the +# Operational copy runs from /var/benchmarks/requant_sweep.sh on the # homeserver. Committed here because the original was never committed, did not # survive the 2026-09-03/04 rebuild, and had to be rewritten from its issue # (#2985). Keep the two in sync by hand. @@ -61,7 +61,7 @@ # --------------------------------------------------------------------------- # Dynamic-quant extension (#3086, round-7 plan §7 R2): IMATRIX= and TENSOR_TYPES= # -# IMATRIX=/mnt-1/training/calib/calib.txt bash requant_sweep.sh +# IMATRIX=/var/training/calib/calib.txt bash requant_sweep.sh # TENSOR_TYPES="re=Q8_0" IMATRIX=... bash requant_sweep.sh # # IMATRIX turns the plain K-quant ladder into the Dynamic-3.0 methodology @@ -86,7 +86,7 @@ # partial writes, not small models.) set -u -BASE=${BASE:-/mnt-1/benchmarks} +BASE=${BASE:-/var/benchmarks} WORK=${WORK:-$BASE/f16work} RESULTS=${RESULTS:-$BASE/1947full} REPO=${REPO:-$BASE/APIARY} diff --git a/analysis/ghidra/benchmarks/corpus/resume_phases.sh b/analysis/ghidra/benchmarks/corpus/resume_phases.sh index a57cf6d41..55f1bcf85 100755 --- a/analysis/ghidra/benchmarks/corpus/resume_phases.sh +++ b/analysis/ghidra/benchmarks/corpus/resume_phases.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash # resume_phases.sh -- #2985: lost in the same rebuild as gptoss_rerun.sh/ # requant_sweep.sh/slots_sweep.sh, never committed before now. Operational -# copy lives at /mnt-1/benchmarks/resume_phases.sh. +# copy lives at /var/benchmarks/resume_phases.sh. # # Status: historical. Written to resume the a99e765 cold sweep after a RAM # install; that sweep was aborted by operator decision on 2026-09-06 (folded @@ -18,14 +18,14 @@ # quant-ladder step it does on the way through. chain2b and chain3 then pick up # on EXTRA_COMPLETE / REQUANT_SWEEP_COMPLETE as before. set -u -cd /mnt-1/benchmarks || exit 1 +cd /var/benchmarks || exit 1 echo "=== $(date -u +%FT%TZ) resume ===" # the pinned head is load-bearing: post-merge main defaults to 17 corpus cases # against the 14 this sweep measured, which would split the matrix down the middle -HEAD=$(git -C /mnt-1/benchmarks/APIARY rev-parse --short HEAD) -BRANCH=$(git -C /mnt-1/benchmarks/APIARY rev-parse --abbrev-ref HEAD) +HEAD=$(git -C /var/benchmarks/APIARY rev-parse --short HEAD) +BRANCH=$(git -C /var/benchmarks/APIARY rev-parse --abbrev-ref HEAD) echo "repo: $HEAD on $BRANCH" if [ "$HEAD" != "a99e765" ]; then echo "ABORT: repo head moved off a99e765. Phases 2-3 must run on the same case" @@ -33,7 +33,7 @@ if [ "$HEAD" != "a99e765" ]; then exit 1 fi -if ! grep -q FULLRUN_COMPLETE /mnt-1/benchmarks/fullrun.log 2>/dev/null; then +if ! grep -q FULLRUN_COMPLETE /var/benchmarks/fullrun.log 2>/dev/null; then echo "ABORT: fullrun.log has no FULLRUN_COMPLETE -- phase 1 is not finished." exit 1 fi @@ -47,29 +47,29 @@ done echo "ollama up; $(docker exec ghidra-ollama-1 ollama list 2>/dev/null | tail -n +2 | wc -l) models local" for s in chain chain2b chain3; do - if pgrep -f "/mnt-1/benchmarks/$s.sh" >/dev/null 2>&1; then + if pgrep -f "/var/benchmarks/$s.sh" >/dev/null 2>&1; then echo "ABORT: $s.sh already running -- refusing to double-start the GPU queue" exit 1 fi done -cd /mnt-1/benchmarks -setsid nohup bash /mnt-1/benchmarks/chain.sh >> /mnt-1/benchmarks/extra.log 2>&1 < /dev/null & +cd /var/benchmarks +setsid nohup bash /var/benchmarks/chain.sh >> /var/benchmarks/extra.log 2>&1 < /dev/null & # 2026-09-05 (#1947 review): only launch a waiter whose downstream script # still exists. gptoss_rerun.sh / requant_sweep.sh / slots_sweep.sh were lost # in the rebuild and were never committed, so chain2b/chain3 would otherwise # sit for 14 days on markers nothing can write and report as "running". -if [ -f /mnt-1/benchmarks/gptoss_rerun.sh ] && [ -f /mnt-1/benchmarks/requant_sweep.sh ]; then - setsid nohup bash /mnt-1/benchmarks/chain2b.sh >> /mnt-1/benchmarks/requant.log 2>&1 < /dev/null & +if [ -f /var/benchmarks/gptoss_rerun.sh ] && [ -f /var/benchmarks/requant_sweep.sh ]; then + setsid nohup bash /var/benchmarks/chain2b.sh >> /var/benchmarks/requant.log 2>&1 < /dev/null & else echo "SKIP chain2b: gptoss_rerun.sh/requant_sweep.sh missing -- phases 2.5/3 not queued" fi -if [ -f /mnt-1/benchmarks/slots_sweep.sh ]; then - setsid nohup bash /mnt-1/benchmarks/chain3.sh >> /mnt-1/benchmarks/slots.log 2>&1 < /dev/null & +if [ -f /var/benchmarks/slots_sweep.sh ]; then + setsid nohup bash /var/benchmarks/chain3.sh >> /var/benchmarks/slots.log 2>&1 < /dev/null & else echo "SKIP chain3: slots_sweep.sh missing -- phase 5 not queued" fi sleep 3 echo "relaunched:" -pgrep -af '/mnt-1/benchmarks/chain' || echo " (nothing came up -- check the logs)" +pgrep -af '/var/benchmarks/chain' || echo " (nothing came up -- check the logs)" echo "phase 2 (49-model extra roster) is now running; requant and slots chain behind it." diff --git a/analysis/ghidra/benchmarks/corpus/round7_build_calib.sh b/analysis/ghidra/benchmarks/corpus/round7_build_calib.sh index fc4357132..94d454031 100755 --- a/analysis/ghidra/benchmarks/corpus/round7_build_calib.sh +++ b/analysis/ghidra/benchmarks/corpus/round7_build_calib.sh @@ -1,9 +1,9 @@ #!/usr/bin/env bash # round7_build_calib.sh -- #3086 R1: build the round-7 imatrix calibration set -# at /mnt-1/training/calib/ from corpus-v1 slices, ready to be handed to +# at /var/training/calib/ from corpus-v1 slices, ready to be handed to # requant_sweep.sh as IMATRIX=... and to export_to_ollama.sh as CALIB_FILE=. # -# Operational copy: /mnt-1/training/round7_build_calib.sh. +# Operational copy: /var/training/round7_build_calib.sh. # # --------------------------------------------------------------------------- # Sources (plan §6.1/§6.4 -- the test set is off limits, all of it): @@ -18,7 +18,7 @@ # EXEC=1; the default only reports what it would do. # sanitised sessions S1 session text, already sanitised through the # production contracts.py path before it lands in -# /mnt-1/training/corpus-v1/ (never leaves the host). +# /var/training/corpus-v1/ (never leaves the host). # REx86 text S4 entries (Zenodo 15420461, CC-BY-4.0; #847's # internal-split check must pass first). # general-text portion S6 CPT shards already pooled by generate_s6.py. @@ -38,13 +38,13 @@ # provenance, token estimate, and the decontamination report. # # Usage: -# ssh homeserver 'bash /mnt-1/training/round7_build_calib.sh' # plan -# ssh homeserver 'EXEC=1 bash /mnt-1/training/round7_build_calib.sh' # build +# ssh homeserver 'bash /var/training/round7_build_calib.sh' # plan +# ssh homeserver 'EXEC=1 bash /var/training/round7_build_calib.sh' # build set -euo pipefail -CALIB=${CALIB:-/mnt-1/training/calib} -CORPUS=${CORPUS:-/mnt-1/training/corpus-v1} -REPO=${REPO:-/mnt-1/benchmarks/APIARY-round7} +CALIB=${CALIB:-/var/training/calib} +CORPUS=${CORPUS:-/var/training/corpus-v1} +REPO=${REPO:-/var/benchmarks/APIARY-round7} EXEC=${EXEC:-0} MIN_TOKENS=${MIN_TOKENS:-2000000} MAX_TOKENS=${MAX_TOKENS:-10000000} diff --git a/analysis/ghidra/benchmarks/corpus/round7_build_corpus.sh b/analysis/ghidra/benchmarks/corpus/round7_build_corpus.sh index af49a0172..b882f5649 100755 --- a/analysis/ghidra/benchmarks/corpus/round7_build_corpus.sh +++ b/analysis/ghidra/benchmarks/corpus/round7_build_corpus.sh @@ -3,11 +3,11 @@ # inside the exact provenance container ci_verify.sh documents # (debian:trixie-slim + the pinned cross toolchains), verified byte-for-byte # against the committed manifest, then copied out to a NEW directory. The -# 14-case corpus at /mnt-1/benchmarks/corpus is left untouched. -# Operational copy: /mnt-1/benchmarks/round7_build_corpus.sh +# 14-case corpus at /var/benchmarks/corpus is left untouched. +# Operational copy: /var/benchmarks/round7_build_corpus.sh set -euo pipefail -REPO=${REPO:-/mnt-1/benchmarks/APIARY-round7} -OUT=${OUT:-/mnt-1/benchmarks/corpus-round7} +REPO=${REPO:-/var/benchmarks/APIARY-round7} +OUT=${OUT:-/var/benchmarks/corpus-round7} NAME=corpus-round7-build docker rm -f "$NAME" >/dev/null 2>&1 || true # no --rm: the built corpus is copied out of the stopped container afterwards, diff --git a/analysis/ghidra/benchmarks/corpus/round7_cache.sh b/analysis/ghidra/benchmarks/corpus/round7_cache.sh index 7cf9a1867..80281ae87 100755 --- a/analysis/ghidra/benchmarks/corpus/round7_cache.sh +++ b/analysis/ghidra/benchmarks/corpus/round7_cache.sh @@ -1,15 +1,15 @@ #!/usr/bin/env bash # round7_cache.sh -- regenerate the Tier B (Ghidra decompilation) cache for the # round-7 pin's 17-case corpus, into its OWN directory. The 14-case cache at -# /mnt-1/benchmarks/tierb-cache stays for the a99e765 clone. +# /var/benchmarks/tierb-cache stays for the a99e765 clone. # -# Operational copy lives at /mnt-1/benchmarks/round7_cache.sh. +# Operational copy lives at /var/benchmarks/round7_cache.sh. # # GHIDRA_VERSION must be exported by hand because the headless service # publishes no version of its own (#2983); the line printed first is the # container's own application.properties so the two can be compared. set -euo pipefail -BASE=${BASE:-/mnt-1/benchmarks} +BASE=${BASE:-/var/benchmarks} REPO=${REPO:-$BASE/APIARY-round7} CORPUS=${CORPUS:-$BASE/corpus-round7} CACHE=${CACHE:-$BASE/tierb-cache-round7} diff --git a/analysis/ghidra/benchmarks/corpus/round7_coldrun.sh b/analysis/ghidra/benchmarks/corpus/round7_coldrun.sh index 5333e8be8..2a8ea1d05 100755 --- a/analysis/ghidra/benchmarks/corpus/round7_coldrun.sh +++ b/analysis/ghidra/benchmarks/corpus/round7_coldrun.sh @@ -4,7 +4,7 @@ # pin -- 17 cases / 79, injection gate v3, pooled-claims-ready transcripts -- # cold slot, live workers stopped, N=2 with automatic 3/5 escalation. # -# Operational copy lives at /mnt-1/benchmarks/round7_coldrun.sh. +# Operational copy lives at /var/benchmarks/round7_coldrun.sh. # # --------------------------------------------------------------------------- # Why this replaced coldrun.sh's a99e765 re-run (operator decision 2026-09-06) @@ -31,7 +31,7 @@ # fails without it, and sweep_extra would still write MODEL_DONE (#2971) # - no other sweep holds the card set -u -BASE=${BASE:-/mnt-1/benchmarks} +BASE=${BASE:-/var/benchmarks} PIN=${PIN:-32dbdeb1} REPO=${REPO:-$BASE/APIARY-round7} OUT=${OUT:-$BASE/round7} diff --git a/analysis/ghidra/benchmarks/corpus/round7_launch.sh b/analysis/ghidra/benchmarks/corpus/round7_launch.sh index 3bea19b89..59195bb03 100644 --- a/analysis/ghidra/benchmarks/corpus/round7_launch.sh +++ b/analysis/ghidra/benchmarks/corpus/round7_launch.sh @@ -2,13 +2,13 @@ # round7_launch.sh -- start the round-7 cold baseline detached, with the VRAM # sampler beside it, after the smoke test has proved the leg scores. # -# Operational copy lives at /mnt-1/benchmarks/round7_launch.sh. +# Operational copy lives at /var/benchmarks/round7_launch.sh. # # Refuses to launch unless round7_smoke.sh has left a Tier B result on this # pin: a Tier-B-only failure is the exact 2026-09-04 defect (#2971), and the # sweep would still write MODEL_DONE over the hole. set -u -BASE=${BASE:-/mnt-1/benchmarks} +BASE=${BASE:-/var/benchmarks} SMOKE=${SMOKE:-$BASE/smoke-round7/tierB_smoke.json} [ -s "$SMOKE" ] || { echo "ABORT: no Tier B smoke result at $SMOKE -- run round7_smoke.sh first"; exit 1; } python3 - "$SMOKE" <<'PY' || exit 1 diff --git a/analysis/ghidra/benchmarks/corpus/round7_prep_pin.sh b/analysis/ghidra/benchmarks/corpus/round7_prep_pin.sh index f97ac3fee..552c616fb 100755 --- a/analysis/ghidra/benchmarks/corpus/round7_prep_pin.sh +++ b/analysis/ghidra/benchmarks/corpus/round7_prep_pin.sh @@ -2,11 +2,11 @@ # prep_round7_clone.sh -- pinned checkout for round 7 (epic #3079): a SECOND # clone, detached at the round-7 pin. The a99e765 clone stays untouched -- its # untracked transcripts are the phase-1/2 evidence and resume_phases.sh guards -# that HEAD. Operational copy: /mnt-1/benchmarks/round7_prep_pin.sh +# that HEAD. Operational copy: /var/benchmarks/round7_prep_pin.sh set -euo pipefail PIN=${PIN:-32dbdeb1face8c8e4791d31a8f4fbbe321e4f6fa} -DST=${DST:-/mnt-1/benchmarks/APIARY-round7} -OLD=${OLD:-/mnt-1/benchmarks/APIARY} +DST=${DST:-/var/benchmarks/APIARY-round7} +OLD=${OLD:-/var/benchmarks/APIARY} url=$(git -C "$OLD" remote get-url origin) if [ -d "$DST/.git" ]; then echo "clone exists: $DST" diff --git a/analysis/ghidra/benchmarks/corpus/round7_prepare.sh b/analysis/ghidra/benchmarks/corpus/round7_prepare.sh index 5f667edd2..7b2d011a2 100644 --- a/analysis/ghidra/benchmarks/corpus/round7_prepare.sh +++ b/analysis/ghidra/benchmarks/corpus/round7_prepare.sh @@ -7,9 +7,9 @@ # 3. round7_cache.sh 17-case Tier B (Ghidra) cache in its own directory # Then run round7_smoke.sh, read its two lines, and only then round7_launch.sh. # -# Operational copy lives at /mnt-1/benchmarks/round7_prepare.sh. +# Operational copy lives at /var/benchmarks/round7_prepare.sh. set -euo pipefail -BASE=${BASE:-/mnt-1/benchmarks} +BASE=${BASE:-/var/benchmarks} log() { echo "$(date -u +%FT%TZ) $*"; } log "step 1/3 pin" bash "$BASE/round7_prep_pin.sh" diff --git a/analysis/ghidra/benchmarks/corpus/round7_smoke.sh b/analysis/ghidra/benchmarks/corpus/round7_smoke.sh index 2b16c572b..52198e0bc 100755 --- a/analysis/ghidra/benchmarks/corpus/round7_smoke.sh +++ b/analysis/ghidra/benchmarks/corpus/round7_smoke.sh @@ -4,9 +4,9 @@ # the pull). One Tier A and one Tier B run of a small local model into a # separate directory that no results glob reads. # -# Operational copy lives at /mnt-1/benchmarks/round7_smoke.sh. +# Operational copy lives at /var/benchmarks/round7_smoke.sh. set -u -BASE=${BASE:-/mnt-1/benchmarks} +BASE=${BASE:-/var/benchmarks} REPO=${REPO:-$BASE/APIARY-round7} OUT=${OUT:-$BASE/smoke-round7} CACHE=${CACHE:-$BASE/tierb-cache-round7} diff --git a/analysis/ghidra/benchmarks/corpus/round7_sweep.sh b/analysis/ghidra/benchmarks/corpus/round7_sweep.sh index ce2b2df97..146b5b73d 100755 --- a/analysis/ghidra/benchmarks/corpus/round7_sweep.sh +++ b/analysis/ghidra/benchmarks/corpus/round7_sweep.sh @@ -6,9 +6,9 @@ # whole point; this only chooses the roster, the output directory and the # regime flags, exactly as round7_coldrun.sh does. # -# Operational copy: /mnt-1/benchmarks/round7_sweep.sh. +# Operational copy: /var/benchmarks/round7_sweep.sh. # -# LIST=/mnt-1/benchmarks/models_round7.txt RESULTS=/mnt-1/benchmarks/round7 \ +# LIST=/var/benchmarks/models_round7.txt RESULTS=/var/benchmarks/round7 \ # STOP_WORKERS=1 bash round7_sweep.sh # # Results land beside the running cold baseline's own output when RESULTS @@ -17,7 +17,7 @@ # rows scored as artefacts land fill in the same matrix the chain aggregates. set -u -BASE=${BASE:-/mnt-1/benchmarks} +BASE=${BASE:-/var/benchmarks} RESULTS=${RESULTS:-$BASE/round7} REPO=${REPO:-$BASE/APIARY-round7} STOP_WORKERS=${STOP_WORKERS:-1} diff --git a/analysis/ghidra/benchmarks/corpus/round7_t0_merge_export.sh b/analysis/ghidra/benchmarks/corpus/round7_t0_merge_export.sh index 0392c5d31..5535d5c5c 100755 --- a/analysis/ghidra/benchmarks/corpus/round7_t0_merge_export.sh +++ b/analysis/ghidra/benchmarks/corpus/round7_t0_merge_export.sh @@ -4,7 +4,7 @@ # convert to GGUF, quantise, and register in Ollama beside its untouched base # twins -- so the training legs (#3083+) hit a proved path instead of a guess. # -# Operational copy lives at /mnt-1/benchmarks/round7_t0_merge_export.sh. +# Operational copy lives at /var/benchmarks/round7_t0_merge_export.sh. # # --------------------------------------------------------------------------- # Why this is CPU-only and why it gates on the cold run @@ -28,8 +28,8 @@ # hard SHA and size above. A mismatch aborts. set -u -BASE=${BASE:-/mnt-1/benchmarks} -RUN=${RUN:-/mnt-1/training/runs/t0-rex86} +BASE=${BASE:-/var/benchmarks} +RUN=${RUN:-/var/training/runs/t0-rex86} # $RUN as hp-unsloth-studio sees it: that container binds /var/training and # RENAMES it to /workspace, so no host path is valid inside it. RUN_C=${RUN_C:-/workspace/runs/t0-rex86} diff --git a/analysis/ghidra/benchmarks/corpus/round7_t0_score.sh b/analysis/ghidra/benchmarks/corpus/round7_t0_score.sh index b6458f1ff..e52cb49c6 100755 --- a/analysis/ghidra/benchmarks/corpus/round7_t0_score.sh +++ b/analysis/ghidra/benchmarks/corpus/round7_t0_score.sh @@ -6,7 +6,7 @@ # pin comparability), the four T0 cases additionally extracted from the run's # transcripts for the per-case read, cold protocol throughout. # -# Operational copy lives at /mnt-1/benchmarks/round7_t0_score.sh. +# Operational copy lives at /var/benchmarks/round7_t0_score.sh. # # --------------------------------------------------------------------------- # Why this file cannot run today and is written but gated @@ -22,7 +22,7 @@ # files) rather than inventing a fourth protocol. # # Nothing here can fire until the positive condition holds: every round-7 T0 -# tag has both tier files or an UNMEASURED marker in /mnt-1/benchmarks/round7/ +# tag has both tier files or an UNMEASURED marker in /var/benchmarks/round7/ # AND no sweep_extra.sh / record_baseline.py / round7_coldrun.sh process is # alive (same shape as chain_cold.sh). # @@ -34,7 +34,7 @@ # per-slice transcript entries after the full-slice run (extract_cases below). set -u -BASE=${BASE:-/mnt-1/benchmarks} +BASE=${BASE:-/var/benchmarks} RESULTS7=${RESULTS7:-$BASE/round7} OUT=${OUT:-$BASE/round7/t0} REPO=${REPO:-$BASE/APIARY-round7} diff --git a/analysis/ghidra/benchmarks/corpus/run_injection_pair.sh b/analysis/ghidra/benchmarks/corpus/run_injection_pair.sh index 98265a1fc..53ae2fff2 100755 --- a/analysis/ghidra/benchmarks/corpus/run_injection_pair.sh +++ b/analysis/ghidra/benchmarks/corpus/run_injection_pair.sh @@ -20,14 +20,14 @@ # MODELS_FILE one Ollama tag per line # OUT_DIR where tier{A,B}__.json land (never inside a # directory another sweep is writing to) -# GHIDRA_CACHE Tier B evidence cache (default /mnt-1/benchmarks/tierb-cache) +# GHIDRA_CACHE Tier B evidence cache (default /var/benchmarks/tierb-cache) # OUTPUT_TOKENS override for this run only (default: the slot's pinned 512; # 23/30 Tier B injection answers hit that cap in #1947, see # #2694 -- 1024 is the recommended value for these cases and # is recorded in every report's qualification_request) set -u MODELS="${1:?models file}"; OUT="${2:?output dir}" -CACHE="${3:-/mnt-1/benchmarks/tierb-cache}" +CACHE="${3:-/var/benchmarks/tierb-cache}" TOKENS="${4:-}" REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/../../../.." && pwd)" CASES="strcpy_note_injected,process_witness_probe,process_and_injection" diff --git a/analysis/ghidra/benchmarks/corpus/shutdown_watcher.sh b/analysis/ghidra/benchmarks/corpus/shutdown_watcher.sh index 05f4dfb5b..4b0f05a66 100755 --- a/analysis/ghidra/benchmarks/corpus/shutdown_watcher.sh +++ b/analysis/ghidra/benchmarks/corpus/shutdown_watcher.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash # shutdown_watcher.sh -- #2985: lost in the same rebuild as gptoss_rerun.sh/ # requant_sweep.sh/slots_sweep.sh, never committed before now. Operational -# copy lives at /mnt-1/benchmarks/shutdown_watcher.sh. +# copy lives at /var/benchmarks/shutdown_watcher.sh. # # Written for the a99e765 sweep's RAM install, which was superseded by the # 2026-09-06 abort (round 7, epic #3079) before it ran -- per project record @@ -21,16 +21,16 @@ # On timeout it does NOT shut down. A stalled sweep is something to look at, not # something to power off underneath. set -u -LOG=/mnt-1/benchmarks/shutdown_watcher.log +LOG=/var/benchmarks/shutdown_watcher.log exec >>"$LOG" 2>&1 echo "=== $(date -u +%FT%TZ) watcher armed; waiting for FULLRUN_COMPLETE ===" DEADLINE=$(( $(date +%s) + 14*3600 )) while :; do - grep -q FULLRUN_COMPLETE /mnt-1/benchmarks/fullrun.log 2>/dev/null && break + grep -q FULLRUN_COMPLETE /var/benchmarks/fullrun.log 2>/dev/null && break if [ "$(date +%s)" -ge "$DEADLINE" ]; then echo "$(date -u +%FT%TZ) TIMEOUT after 14h -- sweep did not complete. NOT shutting down." - touch /mnt-1/benchmarks/WATCHER_TIMED_OUT + touch /var/benchmarks/WATCHER_TIMED_OUT exit 1 fi sleep 60 @@ -38,9 +38,9 @@ done echo "$(date -u +%FT%TZ) FULLRUN_COMPLETE seen" # stop the chain before it launches the extra roster -for pat in '/mnt-1/benchmarks/chain.sh' '/mnt-1/benchmarks/chain2b.sh' 'chain3.sh' \ - '/mnt-1/benchmarks/sweep_extra.sh' '/mnt-1/benchmarks/requant_sweep.sh' \ - '/mnt-1/benchmarks/slots_sweep.sh'; do +for pat in '/var/benchmarks/chain.sh' '/var/benchmarks/chain2b.sh' 'chain3.sh' \ + '/var/benchmarks/sweep_extra.sh' '/var/benchmarks/requant_sweep.sh' \ + '/var/benchmarks/slots_sweep.sh'; do pkill -f "$pat" 2>/dev/null && echo " killed: $pat" done sleep 5 @@ -50,13 +50,13 @@ pgrep -af 'record_baseline|evaluate-models|sweep_extra|requant_sweep|slots_sweep # snapshot what we finished with, so the state is readable after the reboot { echo "phase 1 halted for RAM install at $(date -u +%FT%TZ)" - echo "models MODEL_DONE: $(grep -c MODEL_DONE /mnt-1/benchmarks/fullrun.log)" - echo "result files: $(ls /mnt-1/benchmarks/1947full/*.json 2>/dev/null | wc -l)" - echo " tierA: $(ls /mnt-1/benchmarks/1947full/tierA_*.json 2>/dev/null | wc -l)" - echo " tierB: $(ls /mnt-1/benchmarks/1947full/tierB_*.json 2>/dev/null | wc -l)" - echo "repo head (must stay pinned until phase 3 ends): $(git -C /mnt-1/benchmarks/APIARY rev-parse --short HEAD) on $(git -C /mnt-1/benchmarks/APIARY rev-parse --abbrev-ref HEAD)" -} > /mnt-1/benchmarks/PHASE1_SNAPSHOT.txt -cat /mnt-1/benchmarks/PHASE1_SNAPSHOT.txt + echo "models MODEL_DONE: $(grep -c MODEL_DONE /var/benchmarks/fullrun.log)" + echo "result files: $(ls /var/benchmarks/1947full/*.json 2>/dev/null | wc -l)" + echo " tierA: $(ls /var/benchmarks/1947full/tierA_*.json 2>/dev/null | wc -l)" + echo " tierB: $(ls /var/benchmarks/1947full/tierB_*.json 2>/dev/null | wc -l)" + echo "repo head (must stay pinned until phase 3 ends): $(git -C /var/benchmarks/APIARY rev-parse --short HEAD) on $(git -C /var/benchmarks/APIARY rev-parse --abbrev-ref HEAD)" +} > /var/benchmarks/PHASE1_SNAPSHOT.txt +cat /var/benchmarks/PHASE1_SNAPSHOT.txt # unload the GPU so nothing is mid-write, then stop the stateful containers with a # real timeout -- the default 10s is not enough for Elasticsearch to close cleanly diff --git a/analysis/ghidra/benchmarks/corpus/slots_sweep.sh b/analysis/ghidra/benchmarks/corpus/slots_sweep.sh index b975da76e..5a322b7d9 100755 --- a/analysis/ghidra/benchmarks/corpus/slots_sweep.sh +++ b/analysis/ghidra/benchmarks/corpus/slots_sweep.sh @@ -4,7 +4,7 @@ # sessions,revdeck over a roster. The ghidra slot stays held back (#1795: # Tier B evidence first) -- record_baseline.py owns that axis. # -# Operational copy: /mnt-1/benchmarks/slots_sweep.sh. +# Operational copy: /var/benchmarks/slots_sweep.sh. # # Same regime as the corpus sweeps: cold slot, live workers stopped and # restored by trap, one model loaded at a time, uptime per run recorded. @@ -13,7 +13,7 @@ # is no shared scorer to reuse for these slots). set -u -BASE=${BASE:-/mnt-1/benchmarks} +BASE=${BASE:-/var/benchmarks} OUT=${OUT:-$BASE/round7/slots} LIST=${LIST:-$BASE/models_round7.txt} REPO=${REPO:-$BASE/APIARY-round7} diff --git a/analysis/ghidra/benchmarks/corpus/sweep_extra.sh b/analysis/ghidra/benchmarks/corpus/sweep_extra.sh index cbc5f6e3b..010d13274 100755 --- a/analysis/ghidra/benchmarks/corpus/sweep_extra.sh +++ b/analysis/ghidra/benchmarks/corpus/sweep_extra.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash # Extra-roster sweep: pull -> benchmark -> delete, one model at a time. # -# Operational copy of this file runs from /mnt-1/benchmarks/sweep_extra.sh on +# Operational copy of this file runs from /var/benchmarks/sweep_extra.sh on # the homeserver (its BASE/REPO/LIST/PRESEED/CHECK_NAMES paths below are that # host's layout) -- committed here so a script driving multi-hundred-GB pulls # and real benchmark runs is reviewable and isn't one `rm` away from being @@ -36,10 +36,10 @@ set -u # self-quantized tags and then runs exactly this script over them, so the # self-quant rows are scored by the same code, at the same pin, as every # as-published row they are meant to be compared against. -BASE=${BASE:-/mnt-1/benchmarks/1947full} -REPO=${REPO:-/mnt-1/benchmarks/APIARY} -LIST=${LIST:-/mnt-1/benchmarks/models_extra_all.txt} -PRESEED=${PRESEED:-/mnt-1/benchmarks/preseed.sh} +BASE=${BASE:-/var/benchmarks/1947full} +REPO=${REPO:-/var/benchmarks/APIARY} +LIST=${LIST:-/var/benchmarks/models_extra_all.txt} +PRESEED=${PRESEED:-/var/benchmarks/preseed.sh} MAXTRY=${MAXTRY:-3} # #3023: the cold-slot protocol #2641 established requires that nothing else @@ -63,14 +63,14 @@ KEEP_WEIGHTS_ABOVE_GB=${KEEP_WEIGHTS_ABOVE_GB:-1000} # #3087: round 7 scores on its own pin with its own 17-case Tier B cache and # its own operator tag, so both are overridable; the defaults are the a99e765 # sweep's, unchanged. round7_coldrun.sh sets them. -GHIDRA_CACHE=${GHIDRA_CACHE:-/mnt-1/benchmarks/tierb-cache} +GHIDRA_CACHE=${GHIDRA_CACHE:-/var/benchmarks/tierb-cache} OPERATOR=${OPERATOR:-bg-1947extra} # #2738: fail fast on any roster entry Ollama's client-side hf.co name # validation would reject before a sweep wastes time discovering it -- -# see /mnt-1/benchmarks/oversized-model-aliases.tsv for the bisection and +# see /var/benchmarks/oversized-model-aliases.tsv for the bisection and # the recovery path for an entry that does trip this. -CHECK_NAMES=/mnt-1/benchmarks/check-roster-name-lengths.sh +CHECK_NAMES=/var/benchmarks/check-roster-name-lengths.sh if [ -x "$CHECK_NAMES" ]; then "$CHECK_NAMES" "$LIST" || exit 1 fi @@ -199,7 +199,7 @@ while read -r TAG; do PULLED=0 # -i: Ollama rewrites some quantisation-shaped tags to uppercase on write # (#2738's raven aliases: `ollama create x:q4_k_m` lands as `x:Q4_K_M` -- - # see /mnt-1/benchmarks/oversized-model-aliases.tsv for the measured set), + # see /var/benchmarks/oversized-model-aliases.tsv for the measured set), # so an imported alias may not case-match the roster's own spelling of # $TAG. Match case-insensitively so those entries are recognised as # present. Ollama resolves names case-insensitively itself, so handing the diff --git a/analysis/ghidra/benchmarks/corpus/xortron_next_finalize.sh b/analysis/ghidra/benchmarks/corpus/xortron_next_finalize.sh index f6b3159a7..b12eb8731 100755 --- a/analysis/ghidra/benchmarks/corpus/xortron_next_finalize.sh +++ b/analysis/ghidra/benchmarks/corpus/xortron_next_finalize.sh @@ -35,7 +35,7 @@ # DRY_RUN=1 bash xortron_next_finalize.sh # report only set -u -BASE=${BASE:-/mnt-1/benchmarks} +BASE=${BASE:-/var/benchmarks} RESULTS=${RESULTS:-$BASE/1947full} DRY_RUN=${DRY_RUN:-0} diff --git a/analysis/ghidra/benchmarks/model-quant-benchmark/README.md b/analysis/ghidra/benchmarks/model-quant-benchmark/README.md index 08ad21415..0d6daa84b 100644 --- a/analysis/ghidra/benchmarks/model-quant-benchmark/README.md +++ b/analysis/ghidra/benchmarks/model-quant-benchmark/README.md @@ -38,7 +38,7 @@ instead of re-downloading or re-quantizing anything already on disk. | `rex86_run_all_base.sh` | Orchestrates a whole queue of `rex86_run_base_model.sh` calls (one per candidate model) with proven `-ngl` values already worked out per model/quant -- see its own comments for how those envelope numbers were computed. Edit this file's `run ...` lines to add/remove models from a benchmark round. | | `rex86_backfill_extra_quants.sh` | Fills in quant-level gaps on models that were only ever evaluated at one precision, so the comparison chart is apples-to-apples across every model. Also the reference example for requantizing from an already-quantized GGUF (`--allow-requantize`) when the f16 source has already been cleaned up, instead of re-downloading a multi-hundred-GB snapshot just to fill in one more quant level. | | `gen_answers_md.py