From 7f2a9283ce041e7a0e78af4bbf76d3d9f3efc5ea Mon Sep 17 00:00:00 2001 From: Xore Date: Sun, 6 Sep 2026 10:33:26 +0200 Subject: [PATCH] fix(benchmarks): resolve a tag to Ollama's own spelling, and mark a model whose every run failed Two defects that combined into silent data loss. Three models pulled perfectly, failed all six run attempts, were written MODEL_DONE, and had their weights deleted -- leaving nothing in the results directory to say they had not been measured. ## 1. Ollama's stored tag case is not the case you asked for It rewrites SOME quant-shaped tags and not others. Measured on tags this repo created itself, every one written lowercase: gemma4-26b-a4b-selfquant:q3_k_m <- kept as written gemma4-26b-a4b-selfquant:Q4_K_M <- uppercased by Ollama gemma4-26b-a4b-selfquant:q5_k_m <- kept as written sweep_extra.sh's presence check is `grep -qixF`, case-insensitive, so a folded tag passes it. record_baseline.py then resolves the model by EXACT string match against /api/tags and exits "model tag is not installed". Every run fails for a model that is sitting right there. It hit exactly the rows that matter most: Ornith-35B Q3_K_M and gemma-4-26B-A4B Q3_K_M/Q5_K_M -- the published twins of the self-quantized ladder, i.e. the controls the whole #2245 comparison rests on. gemma Q4_K_M succeeded, which is what made it look like a per-model quirk rather than a systematic one. Fixed in the driver, not in record_baseline.py: that file is on the pinned a99e765 harness, and moving the pin to fix a name lookup would split the scoring vintage for no benefit. The driver now hands the scorer the name Ollama actually holds. Verified live -- the same model that failed three times now scores 62. ## 2. A model that produced nothing left no marker Total failure wrote a GIVEUP line to failures.txt and MODEL_DONE to the log, and nothing at all next to the results. Anything reading the results directory could not tell it from a model that was never in the roster. Same silent-hole class as the false EXTRA_COMPLETE in #3031, and it is why defect 1 went unnoticed until the phase-3 chain refused to advance. Now writes UNMEASURED with the reason, so the state is machine-readable. ## The guard held chain_phase3.sh's completion check -- every roster entry has both tier files OR an explicit marker -- blocked on "roster incomplete (46+3/52)" and would not start the ladder scoring. Without it the ladder and then the 96-model cold run would both have proceeded with three of the most decision-relevant rows missing and nothing announcing it. Refs #1947, #2245, #3031, #2738 --- .../ghidra/benchmarks/corpus/sweep_extra.sh | 33 +++++++++++++++++++ 1 file changed, 33 insertions(+) diff --git a/analysis/ghidra/benchmarks/corpus/sweep_extra.sh b/analysis/ghidra/benchmarks/corpus/sweep_extra.sh index 2826b3021..2c64bea7e 100755 --- a/analysis/ghidra/benchmarks/corpus/sweep_extra.sh +++ b/analysis/ghidra/benchmarks/corpus/sweep_extra.sh @@ -218,8 +218,33 @@ while read -r TAG; do echo "$(date -u +%H:%M:%S) already local: $TAG (will not delete)" fi + # Ollama's stored spelling of a tag is not always the spelling you asked for: + # it rewrites SOME quant-shaped tags and not others. Measured on tags this + # repo created itself, all written lowercase: + # gemma4-26b-a4b-selfquant:q3_k_m <- kept + # gemma4-26b-a4b-selfquant:Q4_K_M <- uppercased by Ollama + # The presence check above is `grep -qixF`, case-insensitive, so it passes -- + # but record_baseline.py resolves a model by EXACT string match against + # /api/tags and exits "model tag is not installed". The result was three runs + # failing, MODEL_DONE written, and the weights deleted, for three models that + # had pulled perfectly (#1947 phase 2: Ornith Q3_K_M and gemma Q3_K_M/Q5_K_M -- + # the published twins of the self-quant ladder, i.e. the most decision-relevant + # rows in the roster). + # + # Fixed here rather than in record_baseline.py on purpose: that file is on the + # pinned a99e765 harness, and moving the pin to fix a name lookup would split + # the scoring vintage for no benefit. Hand the scorer the name Ollama actually + # holds. + RESOLVED=$(docker exec ghidra-ollama-1 ollama list 2>/dev/null | awk '{print $1}' | grep -ixF "$TAG" | head -1) + if [ -n "$RESOLVED" ] && [ "$RESOLVED" != "$TAG" ]; then + echo "$(date -u +%H:%M:%S) tag case differs: roster '$TAG' -> ollama '$RESOLVED'" + TAG="$RESOLVED" + fi + + TIER_OK=0 for tier in A B; do do_run "$tier" "$slug" "$TAG" 1 || continue + TIER_OK=1 do_run "$tier" "$slug" "$TAG" 2 || continue s1=$(score_of "$BASE/tier${tier}_${slug}_run1.json") s2=$(score_of "$BASE/tier${tier}_${slug}_run2.json") @@ -256,6 +281,14 @@ while read -r TAG; do echo "$(date -u +%H:%M:%S) kept $TAG (${free_now}G free, above the ${KEEP_WEIGHTS_ABOVE_GB}G floor)" fi fi + # A model that pulled but produced no result on either tier used to leave only + # a GIVEUP line in failures.txt and a MODEL_DONE in the log -- indistinguishable + # from success to anything reading the results directory. That is the same + # silent-hole class as the false EXTRA_COMPLETE in #3031. + if [ "$TIER_OK" = "0" ] && [ ! -f "$BASE/tierA_${slug}_run1.json" ]; then + mark_unmeasured "$slug" "$TAG" "pulled but every run failed on both tiers" + echo "$(date -u +%H:%M:%S) UNMEASURED $TAG (pulled, no run produced a result)" + fi echo "$(date -u +%H:%M:%S) MODEL_DONE $TAG" done < "$LIST"